edwinbh commited on
Commit
d27a328
·
verified ·
1 Parent(s): 31b205a

Update src/streamlit_app.py

Browse files
Files changed (1) hide show
  1. src/streamlit_app.py +256 -512
src/streamlit_app.py CHANGED
@@ -1,33 +1,24 @@
1
  """
2
- Streamlit Dashboard for DLRM Book Recommendation System
3
- Simple interface for DLRM-based book recommendations
4
  """
5
 
6
  import os
7
  import sys
8
  import streamlit as st
9
 
10
- # Check if CPU_ONLY mode is enabled via command line argument
11
- if len(sys.argv) > 1 and sys.argv[1] == '--cpu-only':
12
- os.environ['CPU_ONLY'] = 'true'
13
- print("🔄 Running in CPU-only mode (CUDA disabled)")
 
 
14
 
15
  import pandas as pd
16
  import numpy as np
17
- import torch
18
- import pickle
19
- from typing import Dict, List, Tuple, Optional
20
  import warnings
21
  warnings.filterwarnings('ignore')
22
 
23
- # Import our DLRM recommender
24
- try:
25
- from dlrm_inference import DLRMBookRecommender, load_dlrm_recommender, TORCHREC_AVAILABLE
26
- except ImportError as e:
27
- print(f"⚠️ Error importing DLRM recommender: {e}")
28
- TORCHREC_AVAILABLE = False
29
-
30
-
31
  # Page configuration
32
  st.set_page_config(
33
  page_title="DLRM Book Recommendations",
@@ -36,9 +27,6 @@ st.set_page_config(
36
  initial_sidebar_state="expanded"
37
  )
38
 
39
- # Check if running in CPU-only mode
40
- cpu_only_mode = os.environ.get('CPU_ONLY', 'false').lower() == 'true'
41
-
42
  # Custom CSS
43
  st.markdown("""
44
  <style>
@@ -48,18 +36,14 @@ st.markdown("""
48
  text-align: center;
49
  margin-bottom: 2rem;
50
  }
51
- .metric-card {
52
- background-color: #f0f2f6;
53
- padding: 1rem;
54
- border-radius: 0.5rem;
55
- border-left: 5px solid #1f77b4;
56
- }
57
- .dlrm-explanation {
58
- background-color: #e8f4fd;
59
- padding: 1rem;
60
  border-radius: 0.5rem;
61
- border-left: 4px solid #0066cc;
62
  margin: 1rem 0;
 
63
  }
64
  .book-card {
65
  background-color: #ffffff;
@@ -68,50 +52,76 @@ st.markdown("""
68
  border: 1px solid #e1e5eb;
69
  margin-bottom: 1rem;
70
  }
71
- .cpu-mode-banner {
72
- background-color: #fff3cd;
73
- color: #856404;
74
- padding: 0.75rem;
75
- border-radius: 0.5rem;
76
- border-left: 4px solid #ffeeba;
77
- margin: 1rem 0;
78
- text-align: center;
79
- }
80
  </style>
81
  """, unsafe_allow_html=True)
82
 
83
  @st.cache_data
84
- def load_data():
85
- """Load and cache the book data"""
86
- try:
87
- books_df = pd.read_csv('Books.csv', encoding='latin-1', low_memory=False)
88
- users_df = pd.read_csv('Users.csv', encoding='latin-1', low_memory=False)
89
- ratings_df = pd.read_csv('Ratings.csv', encoding='latin-1', low_memory=False)
90
-
91
- # Clean column names
92
- books_df.columns = books_df.columns.str.replace('"', '')
93
- users_df.columns = users_df.columns.str.replace('"', '')
94
- ratings_df.columns = ratings_df.columns.str.replace('"', '')
95
-
96
- return books_df, users_df, ratings_df
97
- except Exception as e:
98
- st.error(f"Error loading data: {e}")
99
- return None, None, None
100
-
101
- @st.cache_resource
102
- def load_dlrm_model():
103
- """Load and cache the DLRM model"""
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
104
 
 
 
 
 
105
 
106
- try:
107
- recommender = load_dlrm_recommender("file")
108
- return recommender
109
- except Exception as e:
110
- st.error(f"Error loading DLRM model: {e}")
111
- return None
 
 
 
 
 
 
 
 
 
112
 
113
  def display_book_info(book_isbn, books_df, show_rating=None):
114
- """Display book information with actual book cover"""
115
  book_info = books_df[books_df['ISBN'] == book_isbn]
116
 
117
  if len(book_info) == 0:
@@ -123,26 +133,8 @@ def display_book_info(book_isbn, books_df, show_rating=None):
123
  col1, col2 = st.columns([1, 3])
124
 
125
  with col1:
126
- # Try to display actual book cover from Image-URL-M
127
- image_url = book.get('Image-URL-M', '')
128
-
129
- if image_url and pd.notna(image_url) and str(image_url) != 'nan':
130
- try:
131
- # Clean the URL (sometimes there are issues with Amazon URLs)
132
- clean_url = str(image_url).strip()
133
- if clean_url and 'http' in clean_url:
134
- st.image(clean_url, width=150, caption="📚")
135
- else:
136
- # Fallback to placeholder
137
- st.image("https://via.placeholder.com/150x200?text=📚&color=1f77b4&bg=f0f2f6", width=150)
138
- except Exception as e:
139
- # If image loading fails, show placeholder
140
- st.image("https://via.placeholder.com/150x200?text=📚&color=1f77b4&bg=f0f2f6", width=150)
141
- st.caption("⚠️ Cover unavailable")
142
- else:
143
- # Show placeholder if no image URL
144
- st.image("https://via.placeholder.com/150x200?text=📚&color=1f77b4&bg=f0f2f6", width=150)
145
- st.caption("📚 No cover")
146
 
147
  with col2:
148
  st.markdown(f"**{book['Book-Title']}**")
@@ -159,100 +151,49 @@ def main():
159
  st.markdown('<h1 class="main-header">📚 DLRM Book Recommendation System</h1>', unsafe_allow_html=True)
160
  st.markdown("### Deep Learning Recommendation Model for Personalized Book Suggestions")
161
 
162
- # CPU Mode Banner (if enabled)
163
- if cpu_only_mode:
164
- st.markdown('<div class="cpu-mode-banner">⚙️ Running in CPU-only mode (NVIDIA drivers not required)</div>', unsafe_allow_html=True)
 
 
 
165
 
166
  st.markdown("---")
167
 
168
-
169
- # Load data
170
- with st.spinner("Loading book data..."):
171
- books_df, users_df, ratings_df = load_data()
172
-
173
- if books_df is None:
174
- st.error("Failed to load data. Please check if CSV files are available.")
175
- return
176
 
177
  # Sidebar info
178
- st.sidebar.title("📊 Dataset Information")
179
- st.sidebar.metric("📚 Books", f"{len(books_df):,}")
180
- st.sidebar.metric("👥 Users", f"{len(users_df):,}")
181
- st.sidebar.metric("⭐ Ratings", f"{len(ratings_df):,}")
182
-
183
- # Load DLRM model
184
- with st.spinner("Loading DLRM model..."):
185
- recommender = load_dlrm_model()
186
-
187
- if recommender is None or not hasattr(recommender, 'model') or recommender.model is None:
188
- if cpu_only_mode:
189
- st.warning("⚠️ DLRM model not available in CPU-only mode")
190
- st.info("The app will continue with limited functionality")
191
-
192
- # Show options for browsing books without recommendations
193
- st.subheader("📚 Browse Books")
194
-
195
- # Simple book browser
196
- search_query = st.text_input("Search for books", placeholder="Enter title, author, or publisher")
197
- if search_query:
198
- mask = (
199
- books_df['Book-Title'].str.contains(search_query, case=False, na=False) |
200
- books_df['Book-Author'].str.contains(search_query, case=False, na=False) |
201
- books_df['Publisher'].str.contains(search_query, case=False, na=False)
202
- )
203
- results = books_df[mask].head(20)
204
-
205
- if len(results) > 0:
206
- st.success(f"Found {len(results)} books matching '{search_query}'")
207
- for _, book in results.iterrows():
208
- st.markdown(f"**{book['Book-Title']}** by *{book['Book-Author']}*")
209
- st.write(f"Published: {book.get('Year-Of-Publication', 'Unknown')} | ISBN: {book['ISBN']}")
210
- st.markdown("---")
211
- else:
212
- st.info(f"No books found matching '{search_query}'")
213
-
214
- return
215
- else:
216
- st.error("❌ DLRM model not available")
217
- st.info("Please run the training script first: `python train_dlrm_books.py`")
218
-
219
- st.markdown("### Available Options:")
220
- st.markdown("1. **Train DLRM Model**: Run `python train_dlrm_books.py`")
221
- st.markdown("2. **Prepare Data**: Run `python dlrm_book_recommender.py`")
222
- st.markdown("3. **Check Files**: Ensure preprocessing files exist")
223
- st.markdown("4. **Try CPU-only Mode**: Run `streamlit run streamlit_dlrm_app.py -- --cpu-only`")
224
-
225
- return
226
 
227
- if cpu_only_mode:
228
- st.success("✅ DLRM model loaded successfully in CPU-only mode!")
229
- else:
230
- st.success("✅ DLRM model loaded successfully!")
231
-
232
- # Model info
233
  st.sidebar.markdown("---")
234
- st.sidebar.subheader("🤖 DLRM Model Info")
235
- if recommender.preprocessing_info:
236
- st.sidebar.write(f"Dense features: {len(recommender.dense_cols)}")
237
- st.sidebar.write(f"Categorical features: {len(recommender.cat_cols)}")
238
- st.sidebar.write(f"Embedding dim: 64")
 
 
239
 
240
  # Main interface
241
- tab1, tab2, tab3, tab4 = st.tabs(["🎯 Get Recommendations", "🔍 Test Predictions", "📊 Model Analysis", "📸 Book Gallery"])
242
 
243
  with tab1:
244
- st.header("🎯 DLRM Book Recommendations")
245
- st.info("Get personalized book recommendations using the trained DLRM model")
246
 
247
  # User selection
248
  col1, col2 = st.columns([2, 1])
249
 
250
  with col1:
251
- user_ids = sorted(users_df['User-ID'].unique())
252
- selected_user_id = st.selectbox("Select a user", user_ids[:1000]) # Limit for performance
253
 
254
  with col2:
255
- num_recommendations = st.slider("Number of recommendations", 5, 20, 10)
256
 
257
  # Show user info
258
  user_info = users_df[users_df['User-ID'] == selected_user_id]
@@ -263,391 +204,194 @@ def main():
263
  # User's reading history
264
  user_ratings = ratings_df[ratings_df['User-ID'] == selected_user_id]
265
  if len(user_ratings) > 0:
266
- with st.expander(f"📖 User's Reading History ({len(user_ratings)} books)", expanded=False):
267
- top_rated = user_ratings.sort_values('Book-Rating', ascending=False).head(10)
268
- for _, rating in top_rated.iterrows():
269
  book_info = books_df[books_df['ISBN'] == rating['ISBN']]
270
  if len(book_info) > 0:
271
  book = book_info.iloc[0]
272
  st.write(f"• **{book['Book-Title']}** by {book['Book-Author']} - {rating['Book-Rating']}/10 ⭐")
273
 
274
- if st.button("🚀 Get DLRM Recommendations", type="primary"):
275
- with st.spinner("🤖 DLRM is analyzing user preferences..."):
276
 
277
- # Get candidate books (popular books not rated by user)
278
  user_rated_books = set(user_ratings['ISBN']) if len(user_ratings) > 0 else set()
 
279
 
280
- # Get popular books as candidates
281
- book_popularity = ratings_df.groupby('ISBN').size().sort_values(ascending=False)
282
- candidate_books = [isbn for isbn in book_popularity.head(100).index if isbn not in user_rated_books]
283
-
284
- if len(candidate_books) < num_recommendations:
285
- candidate_books = book_popularity.head(200).index.tolist()
286
 
287
- # Get recommendations
288
- recommendations = recommender.get_user_recommendations(
289
- user_id=selected_user_id,
290
- candidate_books=candidate_books,
291
- k=num_recommendations
292
- )
293
 
294
- if recommendations:
295
- st.success(f"Generated {len(recommendations)} DLRM recommendations!")
296
-
297
- st.subheader("🎯 DLRM Recommendations")
298
-
299
- for i, (book_isbn, score) in enumerate(recommendations, 1):
300
- book_info = books_df[books_df['ISBN'] == book_isbn]
301
- if len(book_info) > 0:
302
- with st.expander(f"{i}. Recommendation (DLRM Score: {score:.4f})", expanded=(i <= 3)):
303
- display_book_info(book_isbn, books_df, show_rating=score)
304
-
305
- # Additional book stats
306
- book_ratings = ratings_df[ratings_df['ISBN'] == book_isbn]
307
- if len(book_ratings) > 0:
308
- avg_rating = book_ratings['Book-Rating'].mean()
309
- num_ratings = len(book_ratings)
310
-
311
- st.markdown('<div class="dlrm-explanation">', unsafe_allow_html=True)
312
- st.markdown("**📊 Book Statistics:**")
313
- st.write(f"Average Rating: {avg_rating:.1f}/10 from {num_ratings} readers")
314
- st.write(f"DLRM Confidence: {score:.1%}")
315
- st.markdown('</div>', unsafe_allow_html=True)
316
- else:
317
- st.write(f"Book with ISBN {book_isbn} not found in database")
318
- else:
319
- st.warning("No recommendations generated")
320
 
321
  with tab2:
322
- st.header("🔍 Test DLRM Predictions")
323
- st.info("Test how well DLRM predicts actual user ratings")
 
 
 
 
 
 
 
324
 
325
  col1, col2 = st.columns(2)
326
 
327
  with col1:
328
- test_user_id = st.selectbox("Select user for testing", user_ids[:500], key="test_user")
 
 
 
 
 
 
 
329
 
330
  with col2:
331
- test_mode = st.radio("Test mode", ["Random books", "User's actual books"])
 
 
 
 
 
 
332
 
333
- if st.button("🧪 Test Predictions", type="secondary"):
334
- with st.spinner("Testing DLRM predictions..."):
335
-
336
- if test_mode == "User's actual books":
337
- # Test on user's actual rated books
338
- user_test_ratings = ratings_df[ratings_df['User-ID'] == test_user_id].sample(min(10, len(user_ratings)))
339
-
340
- if len(user_test_ratings) > 0:
341
- st.subheader("🎯 DLRM vs Actual Ratings")
342
-
343
- predictions = []
344
- actuals = []
345
-
346
- for _, rating in user_test_ratings.iterrows():
347
- book_isbn = rating['ISBN']
348
- actual_rating = rating['Book-Rating']
349
-
350
- # Get DLRM prediction
351
- dlrm_score = recommender.predict_rating(test_user_id, book_isbn)
352
-
353
- predictions.append(dlrm_score)
354
- actuals.append(actual_rating >= 6) # Convert to binary
355
-
356
- # Display comparison
357
- book_info = books_df[books_df['ISBN'] == book_isbn]
358
- if len(book_info) > 0:
359
- book = book_info.iloc[0]
360
-
361
- col1, col2, col3 = st.columns([2, 1, 1])
362
- with col1:
363
- st.write(f"**{book['Book-Title']}**")
364
- st.write(f"*by {book['Book-Author']}*")
365
-
366
- with col2:
367
- st.metric("Actual Rating", f"{actual_rating}/10")
368
-
369
- with col3:
370
- st.metric("DLRM Score", f"{dlrm_score:.3f}")
371
-
372
- # Calculate accuracy
373
- if predictions and actuals:
374
- # Convert DLRM scores to binary predictions
375
- binary_preds = [1 if p > 0.5 else 0 for p in predictions]
376
- accuracy = sum(p == a for p, a in zip(binary_preds, actuals)) / len(actuals)
377
-
378
- st.markdown("---")
379
- st.success(f"🎯 DLRM Accuracy: {accuracy:.1%}")
380
-
381
- # Show correlation
382
- actual_numeric = [rating['Book-Rating'] for _, rating in user_test_ratings.iterrows()]
383
- correlation = np.corrcoef(predictions, actual_numeric)[0, 1] if len(predictions) > 1 else 0
384
- st.info(f"📊 Correlation with actual ratings: {correlation:.3f}")
385
-
386
- else:
387
- st.warning("No ratings found for this user")
388
-
389
- else:
390
- # Test on random books
391
- random_books = books_df.sample(10)['ISBN'].tolist()
392
-
393
- st.subheader("🎲 Random Book Predictions")
394
-
395
- for book_isbn in random_books:
396
- dlrm_score = recommender.predict_rating(test_user_id, book_isbn)
397
-
398
- book_info = books_df[books_df['ISBN'] == book_isbn]
399
- if len(book_info) > 0:
400
- book = book_info.iloc[0]
401
-
402
- col1, col2 = st.columns([3, 1])
403
- with col1:
404
- st.write(f"**{book['Book-Title']}** by *{book['Book-Author']}*")
405
-
406
- with col2:
407
- st.metric("DLRM Score", f"{dlrm_score:.4f}")
408
 
409
  with tab3:
410
- st.header("📊 DLRM Model Analysis")
411
- st.info("Analysis of the DLRM model performance and characteristics")
412
 
413
- # Model architecture info
414
- if recommender and recommender.preprocessing_info:
415
- col1, col2 = st.columns(2)
 
 
 
 
 
 
 
 
 
416
 
417
- with col1:
418
- st.subheader("🏗️ Model Architecture")
419
- st.write(f"**Dense Features ({len(recommender.dense_cols)}):**")
420
- for col in recommender.dense_cols:
421
- st.write(f"• {col}")
422
-
423
- st.write(f"**Categorical Features ({len(recommender.cat_cols)}):**")
424
- for i, col in enumerate(recommender.cat_cols):
425
- st.write(f"• {col}: {recommender.emb_counts[i]} embeddings")
426
 
427
- with col2:
428
- st.subheader("📈 Dataset Statistics")
429
- total_samples = recommender.preprocessing_info.get('total_samples', 0)
430
- positive_rate = recommender.preprocessing_info.get('positive_rate', 0)
431
-
432
- st.metric("Total Samples", f"{total_samples:,}")
433
- st.metric("Positive Rate", f"{positive_rate:.1%}")
434
- st.metric("Train Samples", f"{recommender.preprocessing_info.get('train_samples', 0):,}")
435
- st.metric("Validation Samples", f"{recommender.preprocessing_info.get('val_samples', 0):,}")
436
- st.metric("Test Samples", f"{recommender.preprocessing_info.get('test_samples', 0):,}")
437
 
438
- # Feature importance analysis
439
- st.subheader("🔍 Feature Analysis")
440
 
441
- if st.button("Analyze Feature Importance"):
442
- with st.spinner("Analyzing feature importance..."):
443
-
444
- # Sample some users and books
445
- sample_users = users_df['User-ID'].sample(20).tolist()
446
- sample_books = books_df['ISBN'].sample(20).tolist()
447
-
448
- # Test different feature combinations
449
- st.write("**Feature Impact Analysis:**")
450
 
451
- base_predictions = []
452
- for user_id in sample_users[:5]:
453
- for book_isbn in sample_books[:5]:
454
- score = recommender.predict_rating(user_id, book_isbn)
455
- base_predictions.append(score)
456
 
457
- avg_prediction = np.mean(base_predictions)
458
- st.metric("Average Prediction Score", f"{avg_prediction:.4f}")
459
-
460
- st.success("✅ Feature analysis completed!")
461
-
462
- # Load training results if available
463
- if os.path.exists('dlrm_book_training_results.pkl'):
464
- with open('/home/mr-behdadi/PROJECT/ICE/dlrm_book_training_results.pkl', 'rb') as f:
465
- training_results = pickle.load(f)
466
-
467
- st.subheader("📈 Training Results")
468
-
469
- col1, col2 = st.columns(2)
470
-
471
- with col1:
472
- st.metric("Final Validation AUROC", f"{training_results.get('final_val_auroc', 0):.4f}")
473
- st.metric("Test AUROC", f"{training_results.get('test_auroc', 0):.4f}")
474
-
475
- with col2:
476
- val_history = training_results.get('val_aurocs_history', [])
477
- if val_history:
478
- st.line_chart(pd.DataFrame({
479
- 'Epoch': range(len(val_history)),
480
- 'Validation AUROC': val_history
481
- }).set_index('Epoch'))
482
 
483
- # Instructions
484
  st.markdown("---")
485
  st.markdown("""
486
- ## 🚀 How DLRM Works for Book Recommendations
487
 
488
- **DLRM (Deep Learning Recommendation Model)** is specifically designed for recommendation systems and offers several advantages:
489
 
490
- ### 🏗️ Architecture Benefits:
491
- - **Multi-feature Processing**: Handles both categorical (user ID, book ID, publisher) and numerical (age, ratings) features
492
- - **Embedding Tables**: Learns rich representations for categorical features
493
- - **Cross-feature Interactions**: Captures complex relationships between different features
494
- - **Scalable Design**: Efficiently handles large-scale recommendation datasets
495
 
496
- ### 📊 Features Used:
497
- **Categorical Features:**
498
- - User ID, Book ID, Publisher, Country, Age Group, Publication Decade, Rating Level
 
 
499
 
500
- **Dense Features:**
501
- - Normalized Age, Publication Year, User Activity, Book Popularity, Average Ratings
502
-
503
- ### 🎯 Why DLRM vs LLM for Recommendations:
504
- - **Purpose-built**: Specifically designed for recommendation systems
505
- - **Feature Integration**: Better at combining diverse feature types
506
- - **Scalability**: More efficient for large-scale recommendation tasks
507
- - **Performance**: Higher accuracy for rating prediction tasks
508
- - **Production Ready**: Optimized for real-time inference
509
-
510
- ### 💡 Best Use Cases:
511
- - **Personalized Recommendations**: Based on user behavior and item characteristics
512
- - **Rating Prediction**: Accurately predicts user preferences
513
- - **Cold Start**: Handles new users and items through content features
514
- - **Real-time Serving**: Fast inference for production systems
515
  """)
516
 
517
- with tab4:
518
- st.header("📸 Book Gallery")
519
- st.info("Browse book covers and discover new titles")
520
-
521
- # Gallery options
522
- col1, col2 = st.columns([2, 1])
523
-
524
- with col1:
525
- gallery_mode = st.selectbox(
526
- "Choose gallery mode",
527
- ["Popular Books", "Recent Publications", "Random Selection", "Search Results"]
528
- )
529
-
530
- with col2:
531
- books_per_row = st.slider("Books per row", 2, 6, 4)
532
- max_books = st.slider("Maximum books", 10, 50, 20)
533
-
534
- # Get books based on selected mode
535
- if gallery_mode == "Popular Books":
536
- # Get most rated books
537
- book_popularity = ratings_df.groupby('ISBN').size().sort_values(ascending=False)
538
- gallery_books = books_df[books_df['ISBN'].isin(book_popularity.head(max_books).index)]
539
-
540
- elif gallery_mode == "Recent Publications":
541
- # Get recent books
542
- books_df_temp = books_df.copy()
543
- books_df_temp['Year-Of-Publication'] = pd.to_numeric(books_df_temp['Year-Of-Publication'], errors='coerce')
544
- recent_books = books_df_temp.sort_values('Year-Of-Publication', ascending=False, na_position='last')
545
- gallery_books = recent_books.head(max_books)
546
-
547
- elif gallery_mode == "Random Selection":
548
- # Random books
549
- gallery_books = books_df.sample(min(max_books, len(books_df)))
550
-
551
- else: # Search Results
552
- search_query = st.text_input("Search books for gallery", placeholder="Enter title, author, or publisher")
553
- if search_query:
554
- mask = (
555
- books_df['Book-Title'].str.contains(search_query, case=False, na=False) |
556
- books_df['Book-Author'].str.contains(search_query, case=False, na=False) |
557
- books_df['Publisher'].str.contains(search_query, case=False, na=False)
558
- )
559
- gallery_books = books_df[mask].head(max_books)
560
- else:
561
- gallery_books = books_df.head(max_books)
562
-
563
- # Display gallery
564
- if len(gallery_books) > 0:
565
- st.markdown(f"**📚 Showing {len(gallery_books)} books**")
566
-
567
- # Create grid layout
568
- books_list = gallery_books.to_dict('records')
569
-
570
- # Display books in rows
571
- for i in range(0, len(books_list), books_per_row):
572
- cols = st.columns(books_per_row)
573
-
574
- for j, col in enumerate(cols):
575
- if i + j < len(books_list):
576
- book = books_list[i + j]
577
-
578
- with col:
579
- # Book cover
580
- image_url = book.get('Image-URL-M', '')
581
-
582
- if image_url and pd.notna(image_url) and str(image_url) != 'nan':
583
- try:
584
- clean_url = str(image_url).strip()
585
- if clean_url and 'http' in clean_url:
586
- st.image(clean_url, width='stretch')
587
- else:
588
- st.image("https://via.placeholder.com/150x200?text=📚&color=1f77b4&bg=f0f2f6", width='stretch')
589
- except:
590
- st.image("https://via.placeholder.com/150x200?text=📚&color=1f77b4&bg=f0f2f6", width='stretch')
591
- else:
592
- st.image("https://via.placeholder.com/150x200?text=📚&color=1f77b4&bg=f0f2f6", width='stretch')
593
-
594
- # Book info
595
- title = book['Book-Title']
596
- if len(title) > 40:
597
- title = title[:37] + "..."
598
-
599
- author = book['Book-Author']
600
- if len(author) > 25:
601
- author = author[:22] + "..."
602
-
603
- st.markdown(f"**{title}**")
604
- st.write(f"*{author}*")
605
- st.write(f"📅 {book.get('Year-Of-Publication', 'Unknown')}")
606
-
607
- # Book statistics
608
- book_stats = ratings_df[ratings_df['ISBN'] == book['ISBN']]
609
- if len(book_stats) > 0:
610
- avg_rating = book_stats['Book-Rating'].mean()
611
- num_ratings = len(book_stats)
612
- st.write(f"⭐ {avg_rating:.1f}/10 ({num_ratings} ratings)")
613
- else:
614
- st.write("⭐ No ratings")
615
-
616
- # DLRM prediction button
617
- if recommender and recommender.model:
618
- if st.button(f"🎯 DLRM Score", key=f"dlrm_{book['ISBN']}"):
619
- with st.spinner("Calculating..."):
620
- # Use first user as example
621
- sample_user = users_df['User-ID'].iloc[0]
622
- dlrm_score = recommender.predict_rating(sample_user, book['ISBN'])
623
- st.success(f"DLRM Score: {dlrm_score:.3f}")
624
- else:
625
- st.info("No books found for the selected criteria")
626
-
627
- # Quick stats
628
- st.markdown("---")
629
- st.subheader("📊 Gallery Statistics")
630
-
631
- col1, col2, col3, col4 = st.columns(4)
632
-
633
- with col1:
634
- books_with_covers = sum(1 for _, book in gallery_books.iterrows()
635
- if book.get('Image-URL-M') and pd.notna(book.get('Image-URL-M')))
636
- st.metric("Books with Covers", f"{books_with_covers}/{len(gallery_books)}")
637
-
638
- with col2:
639
- # Convert Year-Of-Publication to numeric, coercing errors to NaN
640
- years = pd.to_numeric(gallery_books['Year-Of-Publication'], errors='coerce')
641
- avg_year = years.mean()
642
- st.metric("Average Publication Year", f"{avg_year:.0f}" if not pd.isna(avg_year) else "Unknown")
643
-
644
- with col3:
645
- unique_authors = gallery_books['Book-Author'].nunique()
646
- st.metric("Unique Authors", unique_authors)
647
-
648
- with col4:
649
- unique_publishers = gallery_books['Publisher'].nunique()
650
- st.metric("Unique Publishers", unique_publishers)
651
-
652
  if __name__ == "__main__":
653
- main()
 
1
  """
2
+ Streamlit Dashboard for DLRM Book Recommendation System - Hugging Face Space Compatible
3
+ Simple interface for DLRM-based book recommendations optimized for HF Spaces
4
  """
5
 
6
  import os
7
  import sys
8
  import streamlit as st
9
 
10
+ # Force CPU-only mode for Hugging Face Spaces
11
+ os.environ['CPU_ONLY'] = 'true'
12
+ os.environ['CUDA_VISIBLE_DEVICES'] = ''
13
+
14
+ # Disable Streamlit telemetry for HF Spaces
15
+ os.environ['STREAMLIT_BROWSER_GATHER_USAGE_STATS'] = 'false'
16
 
17
  import pandas as pd
18
  import numpy as np
 
 
 
19
  import warnings
20
  warnings.filterwarnings('ignore')
21
 
 
 
 
 
 
 
 
 
22
  # Page configuration
23
  st.set_page_config(
24
  page_title="DLRM Book Recommendations",
 
27
  initial_sidebar_state="expanded"
28
  )
29
 
 
 
 
30
  # Custom CSS
31
  st.markdown("""
32
  <style>
 
36
  text-align: center;
37
  margin-bottom: 2rem;
38
  }
39
+ .cpu-mode-banner {
40
+ background-color: #d4edda;
41
+ color: #155724;
42
+ padding: 0.75rem;
 
 
 
 
 
43
  border-radius: 0.5rem;
44
+ border-left: 4px solid #28a745;
45
  margin: 1rem 0;
46
+ text-align: center;
47
  }
48
  .book-card {
49
  background-color: #ffffff;
 
52
  border: 1px solid #e1e5eb;
53
  margin-bottom: 1rem;
54
  }
 
 
 
 
 
 
 
 
 
55
  </style>
56
  """, unsafe_allow_html=True)
57
 
58
  @st.cache_data
59
+ def load_sample_data():
60
+ """Load sample data for demo purposes"""
61
+ # Sample book data
62
+ sample_books = {
63
+ 'ISBN': ['0439023483', '0439358078', '0316666343', '0452264464', '0061120081'],
64
+ 'Book-Title': [
65
+ 'The Hunger Games',
66
+ 'Harry Potter and the Chamber of Secrets',
67
+ 'The Catcher in the Rye',
68
+ '1984',
69
+ 'To Kill a Mockingbird'
70
+ ],
71
+ 'Book-Author': [
72
+ 'Suzanne Collins',
73
+ 'J.K. Rowling',
74
+ 'J.D. Salinger',
75
+ 'George Orwell',
76
+ 'Harper Lee'
77
+ ],
78
+ 'Year-Of-Publication': [2008, 1999, 1951, 1949, 1960],
79
+ 'Publisher': ['Scholastic', 'Scholastic', 'Little, Brown', 'Signet', 'Harper']
80
+ }
81
+
82
+ # Sample users
83
+ sample_users = {
84
+ 'User-ID': [1, 2, 3, 4, 5],
85
+ 'Age': [25, 32, 19, 45, 28],
86
+ 'Location': ['New York, USA', 'London, UK', 'Tokyo, Japan', 'Berlin, Germany', 'Toronto, Canada']
87
+ }
88
+
89
+ # Sample ratings
90
+ sample_ratings = {
91
+ 'User-ID': [1, 1, 2, 2, 3, 3, 4, 4, 5, 5],
92
+ 'ISBN': ['0439023483', '0439358078', '0316666343', '0452264464', '0061120081', '0439023483', '0316666343', '0439358078', '0452264464', '0061120081'],
93
+ 'Book-Rating': [9, 8, 7, 10, 8, 6, 9, 7, 8, 9]
94
+ }
95
+
96
+ books_df = pd.DataFrame(sample_books)
97
+ users_df = pd.DataFrame(sample_users)
98
+ ratings_df = pd.DataFrame(sample_ratings)
99
+
100
+ return books_df, users_df, ratings_df
101
 
102
+ def simulate_dlrm_prediction(user_id, book_isbn, user_data=None, book_data=None):
103
+ """Simulate DLRM prediction for demo purposes"""
104
+ # Simple heuristic-based simulation
105
+ np.random.seed(hash(f"{user_id}_{book_isbn}") % 2**32)
106
 
107
+ base_score = 0.5
108
+
109
+ # User preferences (simulated)
110
+ user_bias = np.random.uniform(-0.2, 0.2)
111
+
112
+ # Book popularity (simulated)
113
+ book_bias = np.random.uniform(-0.1, 0.1)
114
+
115
+ # Add some randomness
116
+ noise = np.random.uniform(-0.05, 0.05)
117
+
118
+ final_score = base_score + user_bias + book_bias + noise
119
+ final_score = max(0.0, min(1.0, final_score)) # Clamp to [0,1]
120
+
121
+ return final_score
122
 
123
  def display_book_info(book_isbn, books_df, show_rating=None):
124
+ """Display book information"""
125
  book_info = books_df[books_df['ISBN'] == book_isbn]
126
 
127
  if len(book_info) == 0:
 
133
  col1, col2 = st.columns([1, 3])
134
 
135
  with col1:
136
+ # Placeholder book cover
137
+ st.image("https://via.placeholder.com/150x200?text=📚&color=1f77b4&bg=f0f2f6", width=150)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
138
 
139
  with col2:
140
  st.markdown(f"**{book['Book-Title']}**")
 
151
  st.markdown('<h1 class="main-header">📚 DLRM Book Recommendation System</h1>', unsafe_allow_html=True)
152
  st.markdown("### Deep Learning Recommendation Model for Personalized Book Suggestions")
153
 
154
+ # HF Space optimized banner
155
+ st.markdown('''
156
+ <div class="cpu-mode-banner">
157
+ 🚀 Optimized for Hugging Face Spaces - CPU-only mode with simulated DLRM predictions
158
+ </div>
159
+ ''', unsafe_allow_html=True)
160
 
161
  st.markdown("---")
162
 
163
+ # Load sample data
164
+ with st.spinner("Loading sample data..."):
165
+ books_df, users_df, ratings_df = load_sample_data()
 
 
 
 
 
166
 
167
  # Sidebar info
168
+ st.sidebar.title("📊 Demo Dataset")
169
+ st.sidebar.metric("📚 Sample Books", len(books_df))
170
+ st.sidebar.metric("👥 Sample Users", len(users_df))
171
+ st.sidebar.metric("⭐ Sample Ratings", len(ratings_df))
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
172
 
 
 
 
 
 
 
173
  st.sidebar.markdown("---")
174
+ st.sidebar.markdown("""
175
+ ### 🔧 HF Space Features:
176
+ - CPU-only processing
177
+ - Simulated DLRM predictions
178
+ - Sample dataset demo
179
+ - No GPU dependencies
180
+ """)
181
 
182
  # Main interface
183
+ tab1, tab2, tab3 = st.tabs(["🎯 Get Recommendations", "📊 How DLRM Works", "🔍 Book Explorer"])
184
 
185
  with tab1:
186
+ st.header("🎯 DLRM Book Recommendations (Simulated)")
187
+ st.info("Demo of DLRM-based recommendations using simulated predictions")
188
 
189
  # User selection
190
  col1, col2 = st.columns([2, 1])
191
 
192
  with col1:
193
+ selected_user_id = st.selectbox("Select a user", users_df['User-ID'].tolist())
 
194
 
195
  with col2:
196
+ num_recommendations = st.slider("Number of recommendations", 3, 5, 5)
197
 
198
  # Show user info
199
  user_info = users_df[users_df['User-ID'] == selected_user_id]
 
204
  # User's reading history
205
  user_ratings = ratings_df[ratings_df['User-ID'] == selected_user_id]
206
  if len(user_ratings) > 0:
207
+ with st.expander(f"📖 User's Reading History ({len(user_ratings)} books)", expanded=True):
208
+ for _, rating in user_ratings.iterrows():
 
209
  book_info = books_df[books_df['ISBN'] == rating['ISBN']]
210
  if len(book_info) > 0:
211
  book = book_info.iloc[0]
212
  st.write(f"• **{book['Book-Title']}** by {book['Book-Author']} - {rating['Book-Rating']}/10 ⭐")
213
 
214
+ if st.button("🚀 Get Simulated DLRM Recommendations", type="primary"):
215
+ with st.spinner("🤖 Simulating DLRM analysis..."):
216
 
217
+ # Get books not rated by user
218
  user_rated_books = set(user_ratings['ISBN']) if len(user_ratings) > 0 else set()
219
+ candidate_books = [isbn for isbn in books_df['ISBN'] if isbn not in user_rated_books]
220
 
221
+ # Get simulated recommendations
222
+ recommendations = []
223
+ for book_isbn in candidate_books:
224
+ score = simulate_dlrm_prediction(selected_user_id, book_isbn)
225
+ recommendations.append((book_isbn, score))
 
226
 
227
+ # Sort and take top recommendations
228
+ recommendations.sort(key=lambda x: x[1], reverse=True)
229
+ recommendations = recommendations[:num_recommendations]
 
 
 
230
 
231
+ st.success(f"Generated {len(recommendations)} simulated DLRM recommendations!")
232
+
233
+ st.subheader("🎯 Simulated DLRM Recommendations")
234
+
235
+ for i, (book_isbn, score) in enumerate(recommendations, 1):
236
+ with st.expander(f"{i}. Recommendation (Simulated DLRM Score: {score:.4f})", expanded=(i <= 2)):
237
+ display_book_info(book_isbn, books_df, show_rating=score)
238
+
239
+ # Additional info
240
+ st.markdown(f"""
241
+ **📊 Prediction Details:**
242
+ - User ID: {selected_user_id}
243
+ - Book ISBN: {book_isbn}
244
+ - Simulated DLRM Confidence: {score:.1%}
245
+ - Recommendation Rank: #{i}
246
+ """)
 
 
 
 
 
 
 
 
 
 
247
 
248
  with tab2:
249
+ st.header("📊 How DLRM Works for Book Recommendations")
250
+
251
+ st.markdown("""
252
+ ## 🤖 Deep Learning Recommendation Model (DLRM)
253
+
254
+ DLRM is specifically designed for recommendation systems and offers several advantages over traditional approaches:
255
+
256
+ ### 🏗️ Architecture Benefits:
257
+ """)
258
 
259
  col1, col2 = st.columns(2)
260
 
261
  with col1:
262
+ st.markdown("""
263
+ **🔧 Technical Features:**
264
+ - Multi-feature processing
265
+ - Embedding tables for categorical features
266
+ - Cross-feature interactions
267
+ - Scalable design for large datasets
268
+ - Real-time inference capability
269
+ """)
270
 
271
  with col2:
272
+ st.markdown("""
273
+ **📊 Input Features:**
274
+ - User ID, Age, Location
275
+ - Book ID, Publisher, Publication Year
276
+ - Rating patterns and user activity
277
+ - Cross-feature interactions
278
+ """)
279
 
280
+ st.markdown("""
281
+ ### 🎯 Why DLRM vs Traditional Methods:
282
+
283
+ | Feature | DLRM | Traditional CF | Content-Based |
284
+ |---------|------|----------------|---------------|
285
+ | **Feature Integration** | ✅ Excellent | ❌ Limited | ⚠️ Moderate |
286
+ | **Cold Start Problem** | ✅ Handles well | ❌ Poor | ✅ Good |
287
+ | **Scalability** | ✅ Highly scalable | ⚠️ Moderate | ✅ Good |
288
+ | **Accuracy** | ✅ High | ⚠️ Moderate | ⚠️ Moderate |
289
+ | **Real-time Inference** | ✅ Fast | ⚠️ Slow | ✅ Fast |
290
+
291
+ ### 💡 Best Use Cases:
292
+ - **E-commerce**: Product recommendations
293
+ - **Streaming**: Content recommendations
294
+ - **Publishing**: Book/article suggestions
295
+ - **Social Media**: Feed optimization
296
+ """)
297
+
298
+ # Demo architecture visualization
299
+ st.subheader("🏗️ DLRM Architecture Overview")
300
+
301
+ st.markdown("""
302
+ ```
303
+ User Features Book Features
304
+ ┌─────────────┐ ┌─────────────┐
305
+ │ User ID │ │ Book ID │
306
+ │ Age Group │ │ Publisher │
307
+ │ Location │ │ Decade │
308
+ └─────────────┘ └─────────────┘
309
+ │ │
310
+ ▼ ▼
311
+ ┌─────────────────────────────┐
312
+ │ Embedding Tables │
313
+ └─────────────────────────────┘
314
+ │
315
+ ▼
316
+ ┌─────────────────────────────┐
317
+ │ Cross-Feature Network │
318
+ └─────────────────────────────┘
319
+ │
320
+ ▼
321
+ ┌─────────────────────────────┐
322
+ │ Rating Prediction │
323
+ │ (0.0 - 1.0 score) │
324
+ └─────────────────────────────┘
325
+ ```
326
+ """)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
327
 
328
  with tab3:
329
+ st.header("🔍 Book Explorer")
330
+ st.info("Browse sample books and see simulated DLRM predictions")
331
 
332
+ # Book selection
333
+ selected_book_isbn = st.selectbox("Select a book", books_df['ISBN'].tolist())
334
+ selected_user_for_prediction = st.selectbox("Select user for prediction", users_df['User-ID'].tolist(), key="pred_user")
335
+
336
+ # Display selected book
337
+ st.subheader("📚 Selected Book")
338
+ display_book_info(selected_book_isbn, books_df)
339
+
340
+ # Show prediction
341
+ if st.button("🎯 Get Simulated DLRM Prediction"):
342
+ with st.spinner("Calculating simulated prediction..."):
343
+ prediction_score = simulate_dlrm_prediction(selected_user_for_prediction, selected_book_isbn)
344
 
345
+ st.success(f"Simulated DLRM Prediction: {prediction_score:.4f}")
 
 
 
 
 
 
 
 
346
 
347
+ # Interpretation
348
+ if prediction_score > 0.7:
349
+ st.success("🎯 High recommendation confidence - User likely to enjoy this book!")
350
+ elif prediction_score > 0.5:
351
+ st.info("⚖️ Moderate recommendation confidence - Could be interesting for user")
352
+ else:
353
+ st.warning("📉 Low recommendation confidence - May not match user preferences")
 
 
 
354
 
355
+ # All books overview
356
+ st.subheader("📚 All Sample Books")
357
 
358
+ for _, book in books_df.iterrows():
359
+ with st.expander(f"{book['Book-Title']} by {book['Book-Author']}"):
360
+ col1, col2 = st.columns([2, 1])
 
 
 
 
 
 
361
 
362
+ with col1:
363
+ st.write(f"**ISBN:** {book['ISBN']}")
364
+ st.write(f"**Publisher:** {book['Publisher']}")
365
+ st.write(f"**Year:** {book['Year-Of-Publication']}")
 
366
 
367
+ with col2:
368
+ # Show ratings from sample users
369
+ book_ratings = ratings_df[ratings_df['ISBN'] == book['ISBN']]
370
+ if len(book_ratings) > 0:
371
+ avg_rating = book_ratings['Book-Rating'].mean()
372
+ st.metric("Avg Rating", f"{avg_rating:.1f}/10")
373
+ st.metric("# Ratings", len(book_ratings))
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
374
 
375
+ # Footer
376
  st.markdown("---")
377
  st.markdown("""
378
+ ### 🚀 About this Demo
379
 
380
+ This is a **Hugging Face Space** compatible version of a DLRM Book Recommendation System:
381
 
382
+ - **CPU-only processing**: No GPU or NVIDIA drivers required
383
+ - **Simulated predictions**: Demonstrates DLRM concept with heuristic-based scoring
384
+ - **Sample dataset**: 5 popular books and 5 sample users
385
+ - **Educational purpose**: Shows how DLRM would work in production
 
386
 
387
+ **For production use:**
388
+ - Train actual DLRM model with PyTorch/TorchRec
389
+ - Use full book datasets (millions of books/users)
390
+ - Deploy on GPU infrastructure for better performance
391
+ - Implement proper feature engineering and preprocessing
392
 
393
+ **🔗 Learn more about DLRM:** [Facebook Research DLRM](https://github.com/facebookresearch/dlrm)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
394
  """)
395
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
396
  if __name__ == "__main__":
397
+ main()