"""
Streamlit Dashboard for DLRM Book Recommendation System - Hugging Face Space Compatible
Simple interface for DLRM-based book recommendations optimized for HF Spaces
"""
import os
import sys
import streamlit as st
# Force CPU-only mode for Hugging Face Spaces
os.environ['CPU_ONLY'] = 'true'
os.environ['CUDA_VISIBLE_DEVICES'] = ''
# Disable Streamlit telemetry for HF Spaces
os.environ['STREAMLIT_BROWSER_GATHER_USAGE_STATS'] = 'false'
import pandas as pd
import numpy as np
import warnings
warnings.filterwarnings('ignore')
# Page configuration
st.set_page_config(
page_title="DLRM Book Recommendations",
page_icon="📚",
layout="wide",
initial_sidebar_state="expanded"
)
# Custom CSS
st.markdown("""
""", unsafe_allow_html=True)
@st.cache_data
def load_sample_data():
"""Load sample data for demo purposes"""
# Sample book data
sample_books = {
'ISBN': ['0439023483', '0439358078', '0316666343', '0452264464', '0061120081'],
'Book-Title': [
'The Hunger Games',
'Harry Potter and the Chamber of Secrets',
'The Catcher in the Rye',
'1984',
'To Kill a Mockingbird'
],
'Book-Author': [
'Suzanne Collins',
'J.K. Rowling',
'J.D. Salinger',
'George Orwell',
'Harper Lee'
],
'Year-Of-Publication': [2008, 1999, 1951, 1949, 1960],
'Publisher': ['Scholastic', 'Scholastic', 'Little, Brown', 'Signet', 'Harper']
}
# Sample users
sample_users = {
'User-ID': [1, 2, 3, 4, 5],
'Age': [25, 32, 19, 45, 28],
'Location': ['New York, USA', 'London, UK', 'Tokyo, Japan', 'Berlin, Germany', 'Toronto, Canada']
}
# Sample ratings
sample_ratings = {
'User-ID': [1, 1, 2, 2, 3, 3, 4, 4, 5, 5],
'ISBN': ['0439023483', '0439358078', '0316666343', '0452264464', '0061120081', '0439023483', '0316666343', '0439358078', '0452264464', '0061120081'],
'Book-Rating': [9, 8, 7, 10, 8, 6, 9, 7, 8, 9]
}
books_df = pd.DataFrame(sample_books)
users_df = pd.DataFrame(sample_users)
ratings_df = pd.DataFrame(sample_ratings)
return books_df, users_df, ratings_df
def simulate_dlrm_prediction(user_id, book_isbn, user_data=None, book_data=None):
"""Simulate DLRM prediction for demo purposes"""
# Simple heuristic-based simulation
np.random.seed(hash(f"{user_id}_{book_isbn}") % 2**32)
base_score = 0.5
# User preferences (simulated)
user_bias = np.random.uniform(-0.2, 0.2)
# Book popularity (simulated)
book_bias = np.random.uniform(-0.1, 0.1)
# Add some randomness
noise = np.random.uniform(-0.05, 0.05)
final_score = base_score + user_bias + book_bias + noise
final_score = max(0.0, min(1.0, final_score)) # Clamp to [0,1]
return final_score
def display_book_info(book_isbn, books_df, show_rating=None):
"""Display book information"""
book_info = books_df[books_df['ISBN'] == book_isbn]
if len(book_info) == 0:
st.write(f"Book with ISBN {book_isbn} not found")
return
book = book_info.iloc[0]
col1, col2 = st.columns([1, 3])
with col1:
# Placeholder book cover
st.image("https://via.placeholder.com/150x200?text=📚&color=1f77b4&bg=f0f2f6", width=150)
with col2:
st.markdown(f"**{book['Book-Title']}**")
st.write(f"*by {book['Book-Author']}*")
st.write(f"📅 Published: {book.get('Year-Of-Publication', 'Unknown')}")
st.write(f"🏢 Publisher: {book.get('Publisher', 'Unknown')}")
st.write(f"📖 ISBN: {book['ISBN']}")
if show_rating is not None:
st.markdown(f"**🎯 DLRM Score: {show_rating:.4f}**")
def main():
# Header
st.markdown('
📚 DLRM Book Recommendation System
', unsafe_allow_html=True)
st.markdown("### Deep Learning Recommendation Model for Personalized Book Suggestions")
# HF Space optimized banner
st.markdown('''
🚀 Optimized for Hugging Face Spaces - CPU-only mode with simulated DLRM predictions
''', unsafe_allow_html=True)
st.markdown("---")
# Load sample data
with st.spinner("Loading sample data..."):
books_df, users_df, ratings_df = load_sample_data()
# Sidebar info
st.sidebar.title("📊 Demo Dataset")
st.sidebar.metric("📚 Sample Books", len(books_df))
st.sidebar.metric("👥 Sample Users", len(users_df))
st.sidebar.metric("⭐ Sample Ratings", len(ratings_df))
st.sidebar.markdown("---")
st.sidebar.markdown("""
### 🔧 HF Space Features:
- CPU-only processing
- Simulated DLRM predictions
- Sample dataset demo
- No GPU dependencies
""")
# Main interface
tab1, tab2, tab3 = st.tabs(["🎯 Get Recommendations", "📊 How DLRM Works", "🔍 Book Explorer"])
with tab1:
st.header("🎯 DLRM Book Recommendations (Simulated)")
st.info("Demo of DLRM-based recommendations using simulated predictions")
# User selection
col1, col2 = st.columns([2, 1])
with col1:
selected_user_id = st.selectbox("Select a user", users_df['User-ID'].tolist())
with col2:
num_recommendations = st.slider("Number of recommendations", 3, 5, 5)
# Show user info
user_info = users_df[users_df['User-ID'] == selected_user_id]
if len(user_info) > 0:
user = user_info.iloc[0]
st.markdown(f"**User Info**: Age: {user.get('Age', 'Unknown')}, Location: {user.get('Location', 'Unknown')}")
# User's reading history
user_ratings = ratings_df[ratings_df['User-ID'] == selected_user_id]
if len(user_ratings) > 0:
with st.expander(f"📖 User's Reading History ({len(user_ratings)} books)", expanded=True):
for _, rating in user_ratings.iterrows():
book_info = books_df[books_df['ISBN'] == rating['ISBN']]
if len(book_info) > 0:
book = book_info.iloc[0]
st.write(f"• **{book['Book-Title']}** by {book['Book-Author']} - {rating['Book-Rating']}/10 ⭐")
if st.button("🚀 Get Simulated DLRM Recommendations", type="primary"):
with st.spinner("🤖 Simulating DLRM analysis..."):
# Get books not rated by user
user_rated_books = set(user_ratings['ISBN']) if len(user_ratings) > 0 else set()
candidate_books = [isbn for isbn in books_df['ISBN'] if isbn not in user_rated_books]
# Get simulated recommendations
recommendations = []
for book_isbn in candidate_books:
score = simulate_dlrm_prediction(selected_user_id, book_isbn)
recommendations.append((book_isbn, score))
# Sort and take top recommendations
recommendations.sort(key=lambda x: x[1], reverse=True)
recommendations = recommendations[:num_recommendations]
st.success(f"Generated {len(recommendations)} simulated DLRM recommendations!")
st.subheader("🎯 Simulated DLRM Recommendations")
for i, (book_isbn, score) in enumerate(recommendations, 1):
with st.expander(f"{i}. Recommendation (Simulated DLRM Score: {score:.4f})", expanded=(i <= 2)):
display_book_info(book_isbn, books_df, show_rating=score)
# Additional info
st.markdown(f"""
**📊 Prediction Details:**
- User ID: {selected_user_id}
- Book ISBN: {book_isbn}
- Simulated DLRM Confidence: {score:.1%}
- Recommendation Rank: #{i}
""")
with tab2:
st.header("📊 How DLRM Works for Book Recommendations")
st.markdown("""
## 🤖 Deep Learning Recommendation Model (DLRM)
DLRM is specifically designed for recommendation systems and offers several advantages over traditional approaches:
### 🏗️ Architecture Benefits:
""")
col1, col2 = st.columns(2)
with col1:
st.markdown("""
**🔧 Technical Features:**
- Multi-feature processing
- Embedding tables for categorical features
- Cross-feature interactions
- Scalable design for large datasets
- Real-time inference capability
""")
with col2:
st.markdown("""
**📊 Input Features:**
- User ID, Age, Location
- Book ID, Publisher, Publication Year
- Rating patterns and user activity
- Cross-feature interactions
""")
st.markdown("""
### 🎯 Why DLRM vs Traditional Methods:
| Feature | DLRM | Traditional CF | Content-Based |
|---------|------|----------------|---------------|
| **Feature Integration** | ✅ Excellent | ❌ Limited | ⚠️ Moderate |
| **Cold Start Problem** | ✅ Handles well | ❌ Poor | ✅ Good |
| **Scalability** | ✅ Highly scalable | ⚠️ Moderate | ✅ Good |
| **Accuracy** | ✅ High | ⚠️ Moderate | ⚠️ Moderate |
| **Real-time Inference** | ✅ Fast | ⚠️ Slow | ✅ Fast |
### 💡 Best Use Cases:
- **E-commerce**: Product recommendations
- **Streaming**: Content recommendations
- **Publishing**: Book/article suggestions
- **Social Media**: Feed optimization
""")
# Demo architecture visualization
st.subheader("🏗️ DLRM Architecture Overview")
st.markdown("""
```
User Features Book Features
┌─────────────┐ ┌─────────────┐
│ User ID │ │ Book ID │
│ Age Group │ │ Publisher │
│ Location │ │ Decade │
└─────────────┘ └─────────────┘
│ │
▼ ▼
┌─────────────────────────────┐
│ Embedding Tables │
└─────────────────────────────┘
│
▼
┌─────────────────────────────┐
│ Cross-Feature Network │
└─────────────────────────────┘
│
▼
┌─────────────────────────────┐
│ Rating Prediction │
│ (0.0 - 1.0 score) │
└─────────────────────────────┘
```
""")
with tab3:
st.header("🔍 Book Explorer")
st.info("Browse sample books and see simulated DLRM predictions")
# Book selection
selected_book_isbn = st.selectbox("Select a book", books_df['ISBN'].tolist())
selected_user_for_prediction = st.selectbox("Select user for prediction", users_df['User-ID'].tolist(), key="pred_user")
# Display selected book
st.subheader("📚 Selected Book")
display_book_info(selected_book_isbn, books_df)
# Show prediction
if st.button("🎯 Get Simulated DLRM Prediction"):
with st.spinner("Calculating simulated prediction..."):
prediction_score = simulate_dlrm_prediction(selected_user_for_prediction, selected_book_isbn)
st.success(f"Simulated DLRM Prediction: {prediction_score:.4f}")
# Interpretation
if prediction_score > 0.7:
st.success("🎯 High recommendation confidence - User likely to enjoy this book!")
elif prediction_score > 0.5:
st.info("⚖️ Moderate recommendation confidence - Could be interesting for user")
else:
st.warning("📉 Low recommendation confidence - May not match user preferences")
# All books overview
st.subheader("📚 All Sample Books")
for _, book in books_df.iterrows():
with st.expander(f"{book['Book-Title']} by {book['Book-Author']}"):
col1, col2 = st.columns([2, 1])
with col1:
st.write(f"**ISBN:** {book['ISBN']}")
st.write(f"**Publisher:** {book['Publisher']}")
st.write(f"**Year:** {book['Year-Of-Publication']}")
with col2:
# Show ratings from sample users
book_ratings = ratings_df[ratings_df['ISBN'] == book['ISBN']]
if len(book_ratings) > 0:
avg_rating = book_ratings['Book-Rating'].mean()
st.metric("Avg Rating", f"{avg_rating:.1f}/10")
st.metric("# Ratings", len(book_ratings))
# Footer
st.markdown("---")
st.markdown("""
### 🚀 About this Demo
This is a **Hugging Face Space** compatible version of a DLRM Book Recommendation System:
- **CPU-only processing**: No GPU or NVIDIA drivers required
- **Simulated predictions**: Demonstrates DLRM concept with heuristic-based scoring
- **Sample dataset**: 5 popular books and 5 sample users
- **Educational purpose**: Shows how DLRM would work in production
**For production use:**
- Train actual DLRM model with PyTorch/TorchRec
- Use full book datasets (millions of books/users)
- Deploy on GPU infrastructure for better performance
- Implement proper feature engineering and preprocessing
**🔗 Learn more about DLRM:** [Facebook Research DLRM](https://github.com/facebookresearch/dlrm)
""")
if __name__ == "__main__":
main()