BookRecommendationSystem / src /streamlit_app.py
edwinbh's picture
Update src/streamlit_app.py
d27a328 verified
Raw History Blame
15.3 kB
"""
Streamlit Dashboard for DLRM Book Recommendation System - Hugging Face Space Compatible
Simple interface for DLRM-based book recommendations optimized for HF Spaces
"""
import os
import sys
import streamlit as st
# Force CPU-only mode for Hugging Face Spaces
os.environ['CPU_ONLY'] = 'true'
os.environ['CUDA_VISIBLE_DEVICES'] = ''
# Disable Streamlit telemetry for HF Spaces
os.environ['STREAMLIT_BROWSER_GATHER_USAGE_STATS'] = 'false'
import pandas as pd
import numpy as np
import warnings
warnings.filterwarnings('ignore')
# Page configuration
st.set_page_config(
page_title="DLRM Book Recommendations",
page_icon="πŸ“š",
layout="wide",
initial_sidebar_state="expanded"
)
# Custom CSS
st.markdown("""
<style>
.main-header {
font-size: 3rem;
color: #1f77b4;
text-align: center;
margin-bottom: 2rem;
}
.cpu-mode-banner {
background-color: #d4edda;
color: #155724;
padding: 0.75rem;
border-radius: 0.5rem;
border-left: 4px solid #28a745;
margin: 1rem 0;
text-align: center;
}
.book-card {
background-color: #ffffff;
padding: 1rem;
border-radius: 0.5rem;
border: 1px solid #e1e5eb;
margin-bottom: 1rem;
}
</style>
""", unsafe_allow_html=True)
@st.cache_data
def load_sample_data():
"""Load sample data for demo purposes"""
# Sample book data
sample_books = {
'ISBN': ['0439023483', '0439358078', '0316666343', '0452264464', '0061120081'],
'Book-Title': [
'The Hunger Games',
'Harry Potter and the Chamber of Secrets',
'The Catcher in the Rye',
'1984',
'To Kill a Mockingbird'
],
'Book-Author': [
'Suzanne Collins',
'J.K. Rowling',
'J.D. Salinger',
'George Orwell',
'Harper Lee'
],
'Year-Of-Publication': [2008, 1999, 1951, 1949, 1960],
'Publisher': ['Scholastic', 'Scholastic', 'Little, Brown', 'Signet', 'Harper']
}
# Sample users
sample_users = {
'User-ID': [1, 2, 3, 4, 5],
'Age': [25, 32, 19, 45, 28],
'Location': ['New York, USA', 'London, UK', 'Tokyo, Japan', 'Berlin, Germany', 'Toronto, Canada']
}
# Sample ratings
sample_ratings = {
'User-ID': [1, 1, 2, 2, 3, 3, 4, 4, 5, 5],
'ISBN': ['0439023483', '0439358078', '0316666343', '0452264464', '0061120081', '0439023483', '0316666343', '0439358078', '0452264464', '0061120081'],
'Book-Rating': [9, 8, 7, 10, 8, 6, 9, 7, 8, 9]
}
books_df = pd.DataFrame(sample_books)
users_df = pd.DataFrame(sample_users)
ratings_df = pd.DataFrame(sample_ratings)
return books_df, users_df, ratings_df
def simulate_dlrm_prediction(user_id, book_isbn, user_data=None, book_data=None):
"""Simulate DLRM prediction for demo purposes"""
# Simple heuristic-based simulation
np.random.seed(hash(f"{user_id}_{book_isbn}") % 2**32)
base_score = 0.5
# User preferences (simulated)
user_bias = np.random.uniform(-0.2, 0.2)
# Book popularity (simulated)
book_bias = np.random.uniform(-0.1, 0.1)
# Add some randomness
noise = np.random.uniform(-0.05, 0.05)
final_score = base_score + user_bias + book_bias + noise
final_score = max(0.0, min(1.0, final_score)) # Clamp to [0,1]
return final_score
def display_book_info(book_isbn, books_df, show_rating=None):
"""Display book information"""
book_info = books_df[books_df['ISBN'] == book_isbn]
if len(book_info) == 0:
st.write(f"Book with ISBN {book_isbn} not found")
return
book = book_info.iloc[0]
col1, col2 = st.columns([1, 3])
with col1:
# Placeholder book cover
st.image("https://via.placeholder.com/150x200?text=πŸ“š&color=1f77b4&bg=f0f2f6", width=150)
with col2:
st.markdown(f"**{book['Book-Title']}**")
st.write(f"*by {book['Book-Author']}*")
st.write(f"πŸ“… Published: {book.get('Year-Of-Publication', 'Unknown')}")
st.write(f"🏒 Publisher: {book.get('Publisher', 'Unknown')}")
st.write(f"πŸ“– ISBN: {book['ISBN']}")
if show_rating is not None:
st.markdown(f"**🎯 DLRM Score: {show_rating:.4f}**")
def main():
# Header
st.markdown('<h1 class="main-header">πŸ“š DLRM Book Recommendation System</h1>', unsafe_allow_html=True)
st.markdown("### Deep Learning Recommendation Model for Personalized Book Suggestions")
# HF Space optimized banner
st.markdown('''
<div class="cpu-mode-banner">
πŸš€ Optimized for Hugging Face Spaces - CPU-only mode with simulated DLRM predictions
</div>
''', unsafe_allow_html=True)
st.markdown("---")
# Load sample data
with st.spinner("Loading sample data..."):
books_df, users_df, ratings_df = load_sample_data()
# Sidebar info
st.sidebar.title("πŸ“Š Demo Dataset")
st.sidebar.metric("πŸ“š Sample Books", len(books_df))
st.sidebar.metric("πŸ‘₯ Sample Users", len(users_df))
st.sidebar.metric("⭐ Sample Ratings", len(ratings_df))
st.sidebar.markdown("---")
st.sidebar.markdown("""
### πŸ”§ HF Space Features:
- CPU-only processing
- Simulated DLRM predictions
- Sample dataset demo
- No GPU dependencies
""")
# Main interface
tab1, tab2, tab3 = st.tabs(["🎯 Get Recommendations", "πŸ“Š How DLRM Works", "πŸ” Book Explorer"])
with tab1:
st.header("🎯 DLRM Book Recommendations (Simulated)")
st.info("Demo of DLRM-based recommendations using simulated predictions")
# User selection
col1, col2 = st.columns([2, 1])
with col1:
selected_user_id = st.selectbox("Select a user", users_df['User-ID'].tolist())
with col2:
num_recommendations = st.slider("Number of recommendations", 3, 5, 5)
# Show user info
user_info = users_df[users_df['User-ID'] == selected_user_id]
if len(user_info) > 0:
user = user_info.iloc[0]
st.markdown(f"**User Info**: Age: {user.get('Age', 'Unknown')}, Location: {user.get('Location', 'Unknown')}")
# User's reading history
user_ratings = ratings_df[ratings_df['User-ID'] == selected_user_id]
if len(user_ratings) > 0:
with st.expander(f"πŸ“– User's Reading History ({len(user_ratings)} books)", expanded=True):
for _, rating in user_ratings.iterrows():
book_info = books_df[books_df['ISBN'] == rating['ISBN']]
if len(book_info) > 0:
book = book_info.iloc[0]
st.write(f"β€’ **{book['Book-Title']}** by {book['Book-Author']} - {rating['Book-Rating']}/10 ⭐")
if st.button("πŸš€ Get Simulated DLRM Recommendations", type="primary"):
with st.spinner("πŸ€– Simulating DLRM analysis..."):
# Get books not rated by user
user_rated_books = set(user_ratings['ISBN']) if len(user_ratings) > 0 else set()
candidate_books = [isbn for isbn in books_df['ISBN'] if isbn not in user_rated_books]
# Get simulated recommendations
recommendations = []
for book_isbn in candidate_books:
score = simulate_dlrm_prediction(selected_user_id, book_isbn)
recommendations.append((book_isbn, score))
# Sort and take top recommendations
recommendations.sort(key=lambda x: x[1], reverse=True)
recommendations = recommendations[:num_recommendations]
st.success(f"Generated {len(recommendations)} simulated DLRM recommendations!")
st.subheader("🎯 Simulated DLRM Recommendations")
for i, (book_isbn, score) in enumerate(recommendations, 1):
with st.expander(f"{i}. Recommendation (Simulated DLRM Score: {score:.4f})", expanded=(i <= 2)):
display_book_info(book_isbn, books_df, show_rating=score)
# Additional info
st.markdown(f"""
**πŸ“Š Prediction Details:**
- User ID: {selected_user_id}
- Book ISBN: {book_isbn}
- Simulated DLRM Confidence: {score:.1%}
- Recommendation Rank: #{i}
""")
with tab2:
st.header("πŸ“Š How DLRM Works for Book Recommendations")
st.markdown("""
## πŸ€– Deep Learning Recommendation Model (DLRM)
DLRM is specifically designed for recommendation systems and offers several advantages over traditional approaches:
### πŸ—οΈ Architecture Benefits:
""")
col1, col2 = st.columns(2)
with col1:
st.markdown("""
**πŸ”§ Technical Features:**
- Multi-feature processing
- Embedding tables for categorical features
- Cross-feature interactions
- Scalable design for large datasets
- Real-time inference capability
""")
with col2:
st.markdown("""
**πŸ“Š Input Features:**
- User ID, Age, Location
- Book ID, Publisher, Publication Year
- Rating patterns and user activity
- Cross-feature interactions
""")
st.markdown("""
### 🎯 Why DLRM vs Traditional Methods:
| Feature | DLRM | Traditional CF | Content-Based |
|---------|------|----------------|---------------|
| **Feature Integration** | βœ… Excellent | ❌ Limited | ⚠️ Moderate |
| **Cold Start Problem** | βœ… Handles well | ❌ Poor | βœ… Good |
| **Scalability** | βœ… Highly scalable | ⚠️ Moderate | βœ… Good |
| **Accuracy** | βœ… High | ⚠️ Moderate | ⚠️ Moderate |
| **Real-time Inference** | βœ… Fast | ⚠️ Slow | βœ… Fast |
### πŸ’‘ Best Use Cases:
- **E-commerce**: Product recommendations
- **Streaming**: Content recommendations
- **Publishing**: Book/article suggestions
- **Social Media**: Feed optimization
""")
# Demo architecture visualization
st.subheader("πŸ—οΈ DLRM Architecture Overview")
st.markdown("""
```
User Features Book Features
β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”
β”‚ User ID β”‚ β”‚ Book ID β”‚
β”‚ Age Group β”‚ β”‚ Publisher β”‚
β”‚ Location β”‚ β”‚ Decade β”‚
β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜
β”‚ β”‚
β–Ό β–Ό
β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”
β”‚ Embedding Tables β”‚
β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜
β”‚
β–Ό
β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”
β”‚ Cross-Feature Network β”‚
β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜
β”‚
β–Ό
β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”
β”‚ Rating Prediction β”‚
β”‚ (0.0 - 1.0 score) β”‚
β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜
```
""")
with tab3:
st.header("πŸ” Book Explorer")
st.info("Browse sample books and see simulated DLRM predictions")
# Book selection
selected_book_isbn = st.selectbox("Select a book", books_df['ISBN'].tolist())
selected_user_for_prediction = st.selectbox("Select user for prediction", users_df['User-ID'].tolist(), key="pred_user")
# Display selected book
st.subheader("πŸ“š Selected Book")
display_book_info(selected_book_isbn, books_df)
# Show prediction
if st.button("🎯 Get Simulated DLRM Prediction"):
with st.spinner("Calculating simulated prediction..."):
prediction_score = simulate_dlrm_prediction(selected_user_for_prediction, selected_book_isbn)
st.success(f"Simulated DLRM Prediction: {prediction_score:.4f}")
# Interpretation
if prediction_score > 0.7:
st.success("🎯 High recommendation confidence - User likely to enjoy this book!")
elif prediction_score > 0.5:
st.info("βš–οΈ Moderate recommendation confidence - Could be interesting for user")
else:
st.warning("πŸ“‰ Low recommendation confidence - May not match user preferences")
# All books overview
st.subheader("πŸ“š All Sample Books")
for _, book in books_df.iterrows():
with st.expander(f"{book['Book-Title']} by {book['Book-Author']}"):
col1, col2 = st.columns([2, 1])
with col1:
st.write(f"**ISBN:** {book['ISBN']}")
st.write(f"**Publisher:** {book['Publisher']}")
st.write(f"**Year:** {book['Year-Of-Publication']}")
with col2:
# Show ratings from sample users
book_ratings = ratings_df[ratings_df['ISBN'] == book['ISBN']]
if len(book_ratings) > 0:
avg_rating = book_ratings['Book-Rating'].mean()
st.metric("Avg Rating", f"{avg_rating:.1f}/10")
st.metric("# Ratings", len(book_ratings))
# Footer
st.markdown("---")
st.markdown("""
### πŸš€ About this Demo
This is a **Hugging Face Space** compatible version of a DLRM Book Recommendation System:
- **CPU-only processing**: No GPU or NVIDIA drivers required
- **Simulated predictions**: Demonstrates DLRM concept with heuristic-based scoring
- **Sample dataset**: 5 popular books and 5 sample users
- **Educational purpose**: Shows how DLRM would work in production
**For production use:**
- Train actual DLRM model with PyTorch/TorchRec
- Use full book datasets (millions of books/users)
- Deploy on GPU infrastructure for better performance
- Implement proper feature engineering and preprocessing
**πŸ”— Learn more about DLRM:** [Facebook Research DLRM](https://github.com/facebookresearch/dlrm)
""")
if __name__ == "__main__":
main()