""" Streamlit Dashboard for DLRM Book Recommendation System - Hugging Face Space Compatible Simple interface for DLRM-based book recommendations optimized for HF Spaces """ import os import sys import streamlit as st # Force CPU-only mode for Hugging Face Spaces os.environ['CPU_ONLY'] = 'true' os.environ['CUDA_VISIBLE_DEVICES'] = '' # Disable Streamlit telemetry for HF Spaces os.environ['STREAMLIT_BROWSER_GATHER_USAGE_STATS'] = 'false' import pandas as pd import numpy as np import warnings warnings.filterwarnings('ignore') # Page configuration st.set_page_config( page_title="DLRM Book Recommendations", page_icon="📚", layout="wide", initial_sidebar_state="expanded" ) # Custom CSS st.markdown(""" """, unsafe_allow_html=True) @st.cache_data def load_sample_data(): """Load sample data for demo purposes""" # Sample book data sample_books = { 'ISBN': ['0439023483', '0439358078', '0316666343', '0452264464', '0061120081'], 'Book-Title': [ 'The Hunger Games', 'Harry Potter and the Chamber of Secrets', 'The Catcher in the Rye', '1984', 'To Kill a Mockingbird' ], 'Book-Author': [ 'Suzanne Collins', 'J.K. Rowling', 'J.D. Salinger', 'George Orwell', 'Harper Lee' ], 'Year-Of-Publication': [2008, 1999, 1951, 1949, 1960], 'Publisher': ['Scholastic', 'Scholastic', 'Little, Brown', 'Signet', 'Harper'] } # Sample users sample_users = { 'User-ID': [1, 2, 3, 4, 5], 'Age': [25, 32, 19, 45, 28], 'Location': ['New York, USA', 'London, UK', 'Tokyo, Japan', 'Berlin, Germany', 'Toronto, Canada'] } # Sample ratings sample_ratings = { 'User-ID': [1, 1, 2, 2, 3, 3, 4, 4, 5, 5], 'ISBN': ['0439023483', '0439358078', '0316666343', '0452264464', '0061120081', '0439023483', '0316666343', '0439358078', '0452264464', '0061120081'], 'Book-Rating': [9, 8, 7, 10, 8, 6, 9, 7, 8, 9] } books_df = pd.DataFrame(sample_books) users_df = pd.DataFrame(sample_users) ratings_df = pd.DataFrame(sample_ratings) return books_df, users_df, ratings_df def simulate_dlrm_prediction(user_id, book_isbn, user_data=None, book_data=None): """Simulate DLRM prediction for demo purposes""" # Simple heuristic-based simulation np.random.seed(hash(f"{user_id}_{book_isbn}") % 2**32) base_score = 0.5 # User preferences (simulated) user_bias = np.random.uniform(-0.2, 0.2) # Book popularity (simulated) book_bias = np.random.uniform(-0.1, 0.1) # Add some randomness noise = np.random.uniform(-0.05, 0.05) final_score = base_score + user_bias + book_bias + noise final_score = max(0.0, min(1.0, final_score)) # Clamp to [0,1] return final_score def display_book_info(book_isbn, books_df, show_rating=None): """Display book information""" book_info = books_df[books_df['ISBN'] == book_isbn] if len(book_info) == 0: st.write(f"Book with ISBN {book_isbn} not found") return book = book_info.iloc[0] col1, col2 = st.columns([1, 3]) with col1: # Placeholder book cover st.image("https://via.placeholder.com/150x200?text=📚&color=1f77b4&bg=f0f2f6", width=150) with col2: st.markdown(f"**{book['Book-Title']}**") st.write(f"*by {book['Book-Author']}*") st.write(f"📅 Published: {book.get('Year-Of-Publication', 'Unknown')}") st.write(f"🏢 Publisher: {book.get('Publisher', 'Unknown')}") st.write(f"📖 ISBN: {book['ISBN']}") if show_rating is not None: st.markdown(f"**🎯 DLRM Score: {show_rating:.4f}**") def main(): # Header st.markdown('

📚 DLRM Book Recommendation System

', unsafe_allow_html=True) st.markdown("### Deep Learning Recommendation Model for Personalized Book Suggestions") # HF Space optimized banner st.markdown('''
🚀 Optimized for Hugging Face Spaces - CPU-only mode with simulated DLRM predictions
''', unsafe_allow_html=True) st.markdown("---") # Load sample data with st.spinner("Loading sample data..."): books_df, users_df, ratings_df = load_sample_data() # Sidebar info st.sidebar.title("📊 Demo Dataset") st.sidebar.metric("📚 Sample Books", len(books_df)) st.sidebar.metric("👥 Sample Users", len(users_df)) st.sidebar.metric("⭐ Sample Ratings", len(ratings_df)) st.sidebar.markdown("---") st.sidebar.markdown(""" ### 🔧 HF Space Features: - CPU-only processing - Simulated DLRM predictions - Sample dataset demo - No GPU dependencies """) # Main interface tab1, tab2, tab3 = st.tabs(["🎯 Get Recommendations", "📊 How DLRM Works", "🔍 Book Explorer"]) with tab1: st.header("🎯 DLRM Book Recommendations (Simulated)") st.info("Demo of DLRM-based recommendations using simulated predictions") # User selection col1, col2 = st.columns([2, 1]) with col1: selected_user_id = st.selectbox("Select a user", users_df['User-ID'].tolist()) with col2: num_recommendations = st.slider("Number of recommendations", 3, 5, 5) # Show user info user_info = users_df[users_df['User-ID'] == selected_user_id] if len(user_info) > 0: user = user_info.iloc[0] st.markdown(f"**User Info**: Age: {user.get('Age', 'Unknown')}, Location: {user.get('Location', 'Unknown')}") # User's reading history user_ratings = ratings_df[ratings_df['User-ID'] == selected_user_id] if len(user_ratings) > 0: with st.expander(f"📖 User's Reading History ({len(user_ratings)} books)", expanded=True): for _, rating in user_ratings.iterrows(): book_info = books_df[books_df['ISBN'] == rating['ISBN']] if len(book_info) > 0: book = book_info.iloc[0] st.write(f"• **{book['Book-Title']}** by {book['Book-Author']} - {rating['Book-Rating']}/10 ⭐") if st.button("🚀 Get Simulated DLRM Recommendations", type="primary"): with st.spinner("🤖 Simulating DLRM analysis..."): # Get books not rated by user user_rated_books = set(user_ratings['ISBN']) if len(user_ratings) > 0 else set() candidate_books = [isbn for isbn in books_df['ISBN'] if isbn not in user_rated_books] # Get simulated recommendations recommendations = [] for book_isbn in candidate_books: score = simulate_dlrm_prediction(selected_user_id, book_isbn) recommendations.append((book_isbn, score)) # Sort and take top recommendations recommendations.sort(key=lambda x: x[1], reverse=True) recommendations = recommendations[:num_recommendations] st.success(f"Generated {len(recommendations)} simulated DLRM recommendations!") st.subheader("🎯 Simulated DLRM Recommendations") for i, (book_isbn, score) in enumerate(recommendations, 1): with st.expander(f"{i}. Recommendation (Simulated DLRM Score: {score:.4f})", expanded=(i <= 2)): display_book_info(book_isbn, books_df, show_rating=score) # Additional info st.markdown(f""" **📊 Prediction Details:** - User ID: {selected_user_id} - Book ISBN: {book_isbn} - Simulated DLRM Confidence: {score:.1%} - Recommendation Rank: #{i} """) with tab2: st.header("📊 How DLRM Works for Book Recommendations") st.markdown(""" ## 🤖 Deep Learning Recommendation Model (DLRM) DLRM is specifically designed for recommendation systems and offers several advantages over traditional approaches: ### 🏗️ Architecture Benefits: """) col1, col2 = st.columns(2) with col1: st.markdown(""" **🔧 Technical Features:** - Multi-feature processing - Embedding tables for categorical features - Cross-feature interactions - Scalable design for large datasets - Real-time inference capability """) with col2: st.markdown(""" **📊 Input Features:** - User ID, Age, Location - Book ID, Publisher, Publication Year - Rating patterns and user activity - Cross-feature interactions """) st.markdown(""" ### 🎯 Why DLRM vs Traditional Methods: | Feature | DLRM | Traditional CF | Content-Based | |---------|------|----------------|---------------| | **Feature Integration** | ✅ Excellent | ❌ Limited | ⚠️ Moderate | | **Cold Start Problem** | ✅ Handles well | ❌ Poor | ✅ Good | | **Scalability** | ✅ Highly scalable | ⚠️ Moderate | ✅ Good | | **Accuracy** | ✅ High | ⚠️ Moderate | ⚠️ Moderate | | **Real-time Inference** | ✅ Fast | ⚠️ Slow | ✅ Fast | ### 💡 Best Use Cases: - **E-commerce**: Product recommendations - **Streaming**: Content recommendations - **Publishing**: Book/article suggestions - **Social Media**: Feed optimization """) # Demo architecture visualization st.subheader("🏗️ DLRM Architecture Overview") st.markdown(""" ``` User Features Book Features ┌─────────────┐ ┌─────────────┐ │ User ID │ │ Book ID │ │ Age Group │ │ Publisher │ │ Location │ │ Decade │ └─────────────┘ └─────────────┘ │ │ ▼ ▼ ┌─────────────────────────────┐ │ Embedding Tables │ └─────────────────────────────┘ │ ▼ ┌─────────────────────────────┐ │ Cross-Feature Network │ └─────────────────────────────┘ │ ▼ ┌─────────────────────────────┐ │ Rating Prediction │ │ (0.0 - 1.0 score) │ └─────────────────────────────┘ ``` """) with tab3: st.header("🔍 Book Explorer") st.info("Browse sample books and see simulated DLRM predictions") # Book selection selected_book_isbn = st.selectbox("Select a book", books_df['ISBN'].tolist()) selected_user_for_prediction = st.selectbox("Select user for prediction", users_df['User-ID'].tolist(), key="pred_user") # Display selected book st.subheader("📚 Selected Book") display_book_info(selected_book_isbn, books_df) # Show prediction if st.button("🎯 Get Simulated DLRM Prediction"): with st.spinner("Calculating simulated prediction..."): prediction_score = simulate_dlrm_prediction(selected_user_for_prediction, selected_book_isbn) st.success(f"Simulated DLRM Prediction: {prediction_score:.4f}") # Interpretation if prediction_score > 0.7: st.success("🎯 High recommendation confidence - User likely to enjoy this book!") elif prediction_score > 0.5: st.info("⚖️ Moderate recommendation confidence - Could be interesting for user") else: st.warning("📉 Low recommendation confidence - May not match user preferences") # All books overview st.subheader("📚 All Sample Books") for _, book in books_df.iterrows(): with st.expander(f"{book['Book-Title']} by {book['Book-Author']}"): col1, col2 = st.columns([2, 1]) with col1: st.write(f"**ISBN:** {book['ISBN']}") st.write(f"**Publisher:** {book['Publisher']}") st.write(f"**Year:** {book['Year-Of-Publication']}") with col2: # Show ratings from sample users book_ratings = ratings_df[ratings_df['ISBN'] == book['ISBN']] if len(book_ratings) > 0: avg_rating = book_ratings['Book-Rating'].mean() st.metric("Avg Rating", f"{avg_rating:.1f}/10") st.metric("# Ratings", len(book_ratings)) # Footer st.markdown("---") st.markdown(""" ### 🚀 About this Demo This is a **Hugging Face Space** compatible version of a DLRM Book Recommendation System: - **CPU-only processing**: No GPU or NVIDIA drivers required - **Simulated predictions**: Demonstrates DLRM concept with heuristic-based scoring - **Sample dataset**: 5 popular books and 5 sample users - **Educational purpose**: Shows how DLRM would work in production **For production use:** - Train actual DLRM model with PyTorch/TorchRec - Use full book datasets (millions of books/users) - Deploy on GPU infrastructure for better performance - Implement proper feature engineering and preprocessing **🔗 Learn more about DLRM:** [Facebook Research DLRM](https://github.com/facebookresearch/dlrm) """) if __name__ == "__main__": main()