Maaz commited on
Commit
559b281
·
verified ·
1 Parent(s): 6839b67

Upload 22 files

Browse files
.gitattributes CHANGED
@@ -1,35 +1,4 @@
1
- *.7z filter=lfs diff=lfs merge=lfs -text
2
- *.arrow filter=lfs diff=lfs merge=lfs -text
3
- *.bin filter=lfs diff=lfs merge=lfs -text
4
- *.bz2 filter=lfs diff=lfs merge=lfs -text
5
- *.ckpt filter=lfs diff=lfs merge=lfs -text
6
- *.ftz filter=lfs diff=lfs merge=lfs -text
7
- *.gz filter=lfs diff=lfs merge=lfs -text
8
- *.h5 filter=lfs diff=lfs merge=lfs -text
9
- *.joblib filter=lfs diff=lfs merge=lfs -text
10
- *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
- *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
- *.model filter=lfs diff=lfs merge=lfs -text
13
- *.msgpack filter=lfs diff=lfs merge=lfs -text
14
- *.npy filter=lfs diff=lfs merge=lfs -text
15
- *.npz filter=lfs diff=lfs merge=lfs -text
16
- *.onnx filter=lfs diff=lfs merge=lfs -text
17
- *.ot filter=lfs diff=lfs merge=lfs -text
18
- *.parquet filter=lfs diff=lfs merge=lfs -text
19
- *.pb filter=lfs diff=lfs merge=lfs -text
20
- *.pickle filter=lfs diff=lfs merge=lfs -text
21
- *.pkl filter=lfs diff=lfs merge=lfs -text
22
- *.pt filter=lfs diff=lfs merge=lfs -text
23
- *.pth filter=lfs diff=lfs merge=lfs -text
24
- *.rar filter=lfs diff=lfs merge=lfs -text
25
- *.safetensors filter=lfs diff=lfs merge=lfs -text
26
- saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
- *.tar.* filter=lfs diff=lfs merge=lfs -text
28
- *.tar filter=lfs diff=lfs merge=lfs -text
29
- *.tflite filter=lfs diff=lfs merge=lfs -text
30
- *.tgz filter=lfs diff=lfs merge=lfs -text
31
- *.wasm filter=lfs diff=lfs merge=lfs -text
32
- *.xz filter=lfs diff=lfs merge=lfs -text
33
- *.zip filter=lfs diff=lfs merge=lfs -text
34
- *.zst filter=lfs diff=lfs merge=lfs -text
35
- *tfevents* filter=lfs diff=lfs merge=lfs -text
 
1
+ src/trained_model/similarity_matrix.pkl filter=lfs diff=lfs merge=lfs -text
2
+ *.ipynb linguist-detectable=false
3
+ data/credits.csv filter=lfs diff=lfs merge=lfs -text
4
+ src/trained_model/processed_data.pkl filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
.github/workflows/python-app.yml ADDED
@@ -0,0 +1,87 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # This workflow will install Python dependencies, run tests and lint with a single version of Python
2
+ # For more information see: https://docs.github.com/en/actions/automating-builds-and-tests/building-and-testing-python
3
+
4
+ name: Python application
5
+
6
+ on:
7
+ push:
8
+ branches: [ "main" ]
9
+ pull_request:
10
+ branches: [ "main" ]
11
+
12
+ permissions:
13
+ contents: read
14
+
15
+ jobs:
16
+ build:
17
+ runs-on: ubuntu-latest
18
+ steps:
19
+ - uses: actions/checkout@v4
20
+
21
+ - name: Set up Python 3.10
22
+ uses: actions/setup-python@v3
23
+ with:
24
+ python-version: "3.10"
25
+
26
+ - name: Install dependencies
27
+ run: |
28
+ python -m pip install --upgrade pip setuptools wheel
29
+ pip install flake8 pytest pytest-flask pandas scikit-learn flask fuzzywuzzy python-Levenshtein
30
+ if [ -f requirements.txt ]; then pip install -r requirements.txt; fi
31
+
32
+ - name: Verify data and model directories
33
+ run: |
34
+ mkdir -p data/processed_dataset
35
+ mkdir -p src/trained_model
36
+ touch data/movies.csv data/credits.csv
37
+ if [ ! -d "src/trained_model" ]; then
38
+ echo "Model directory missing!"
39
+ exit 1
40
+ fi
41
+ if [ ! -d "src/templates" ] || [ ! -f "src/templates/index.html" ]; then
42
+ echo "Template files missing!"
43
+ exit 1
44
+ fi
45
+
46
+ - name: Lint with flake8
47
+ run: |
48
+ flake8 . --count --select=E9,F63,F7,F82 --show-source --statistics
49
+ flake8 . --count --exit-zero --max-complexity=10 --max-line-length=127 --statistics
50
+
51
+ - name: Process data and train model
52
+ run: |
53
+ python src/app/main.py
54
+
55
+ - name: Verify model artifacts
56
+ run: |
57
+ if [ ! -f "src/trained_model/processed_data.pkl" ] || \
58
+ [ ! -f "src/trained_model/similarity_matrix.pkl" ] || \
59
+ [ ! -f "src/trained_model/vectorizer.pkl" ]; then
60
+ echo "Model artifacts missing!"
61
+ exit 1
62
+ fi
63
+
64
+ - name: Test with pytest
65
+ run: |
66
+ python -m pytest tests/ -v
67
+
68
+ - name: Start and Test Flask App
69
+ run: |
70
+ python src/app/app.py &
71
+ sleep 10
72
+ curl --retry 5 --retry-delay 5 --retry-connrefused http://127.0.0.1:5000/ || exit 1
73
+ pkill -f "python src/app/app.py"
74
+ env:
75
+ FLASK_ENV: testing
76
+ FLASK_DEBUG: 0
77
+
78
+
79
+ - name: Check setup.py
80
+ run: |
81
+ if [ -f setup.py ]; then
82
+ python setup.py check
83
+ python setup.py sdist bdist_wheel
84
+ pip install -e .
85
+ else
86
+ echo "setup.py not found - skipping package build"
87
+ fi
.gitignore ADDED
@@ -0,0 +1,171 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Byte-compiled / optimized / DLL files
2
+ __pycache__/
3
+ *.py[cod]
4
+ *$py.class
5
+
6
+ # C extensions
7
+ *.so
8
+
9
+ # Distribution / packaging
10
+ .Python
11
+ build/
12
+ develop-eggs/
13
+ dist/
14
+ downloads/
15
+ eggs/
16
+ .eggs/
17
+ lib/
18
+ lib64/
19
+ parts/
20
+ sdist/
21
+ var/
22
+ wheels/
23
+ share/python-wheels/
24
+ *.egg-info/
25
+ .installed.cfg
26
+ *.egg
27
+ MANIFEST
28
+
29
+ # PyInstaller
30
+ # Usually these files are written by a python script from a template
31
+ # before PyInstaller builds the exe, so as to inject date/other infos into it.
32
+ *.manifest
33
+ *.spec
34
+
35
+ # Installer logs
36
+ pip-log.txt
37
+ pip-delete-this-directory.txt
38
+
39
+ # Unit test / coverage reports
40
+ htmlcov/
41
+ .tox/
42
+ .nox/
43
+ .coverage
44
+ .coverage.*
45
+ .cache
46
+ nosetests.xml
47
+ coverage.xml
48
+ *.cover
49
+ *.py,cover
50
+ .hypothesis/
51
+ .pytest_cache/
52
+ cover/
53
+
54
+ # Translations
55
+ *.mo
56
+ *.pot
57
+
58
+ # Django stuff:
59
+ *.log
60
+ local_settings.py
61
+ db.sqlite3
62
+ db.sqlite3-journal
63
+
64
+ # Flask stuff:
65
+ instance/
66
+ .webassets-cache
67
+
68
+ # Scrapy stuff:
69
+ .scrapy
70
+
71
+ # Sphinx documentation
72
+ docs/_build/
73
+
74
+ # PyBuilder
75
+ .pybuilder/
76
+ target/
77
+
78
+ # Jupyter Notebook
79
+ .ipynb_checkpoints
80
+
81
+ # IPython
82
+ profile_default/
83
+ ipython_config.py
84
+
85
+ # pyenv
86
+ # For a library or package, you might want to ignore these files since the code is
87
+ # intended to run in multiple environments; otherwise, check them in:
88
+ # .python-version
89
+
90
+ # pipenv
91
+ # According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
92
+ # However, in case of collaboration, if having platform-specific dependencies or dependencies
93
+ # having no cross-platform support, pipenv may install dependencies that don't work, or not
94
+ # install all needed dependencies.
95
+ #Pipfile.lock
96
+
97
+ # UV
98
+ # Similar to Pipfile.lock, it is generally recommended to include uv.lock in version control.
99
+ # This is especially recommended for binary packages to ensure reproducibility, and is more
100
+ # commonly ignored for libraries.
101
+ #uv.lock
102
+
103
+ # poetry
104
+ # Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
105
+ # This is especially recommended for binary packages to ensure reproducibility, and is more
106
+ # commonly ignored for libraries.
107
+ # https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
108
+ #poetry.lock
109
+
110
+ # pdm
111
+ # Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
112
+ #pdm.lock
113
+ # pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
114
+ # in version control.
115
+ # https://pdm.fming.dev/latest/usage/project/#working-with-version-control
116
+ .pdm.toml
117
+ .pdm-python
118
+ .pdm-build/
119
+
120
+ # PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
121
+ __pypackages__/
122
+
123
+ # Celery stuff
124
+ celerybeat-schedule
125
+ celerybeat.pid
126
+
127
+ # SageMath parsed files
128
+ *.sage.py
129
+
130
+ # Environments
131
+ .env
132
+ .venv
133
+ env/
134
+ venv/
135
+ ENV/
136
+ env.bak/
137
+ venv.bak/
138
+
139
+ # Spyder project settings
140
+ .spyderproject
141
+ .spyproject
142
+
143
+ # Rope project settings
144
+ .ropeproject
145
+
146
+ # mkdocs documentation
147
+ /site
148
+
149
+ # mypy
150
+ .mypy_cache/
151
+ .dmypy.json
152
+ dmypy.json
153
+
154
+ # Pyre type checker
155
+ .pyre/
156
+
157
+ # pytype static type analyzer
158
+ .pytype/
159
+
160
+ # Cython debug symbols
161
+ cython_debug/
162
+
163
+ # PyCharm
164
+ # JetBrains specific template is maintained in a separate JetBrains.gitignore that can
165
+ # be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
166
+ # and can be added to the global gitignore or merged into this file. For a more nuclear
167
+ # option (not recommended) you can uncomment the following to ignore the entire idea folder.
168
+ #.idea/
169
+
170
+ # PyPI configuration file
171
+ .pypirc
README.md CHANGED
@@ -1,10 +1,93 @@
1
- ---
2
- title: Test2
3
- emoji: 🏃
4
- colorFrom: gray
5
- colorTo: blue
6
- sdk: docker
7
- pinned: false
8
- ---
9
-
10
- Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Movie Recommendation System
2
+
3
+ ## 🎥 Watch the Demo👇
4
+
5
+ ![Watch on YouTube|Movie_recommendation_system](https://www.youtube.com/watch?v=FWnURzv_P0Q)
6
+
7
+ ## Overview
8
+ The **Movie Recommendation System** is a Python-based project designed to recommend movies to users based on content similarity. Using natural language processing and machine learning techniques, this system analyzes movie metadata to provide tailored recommendations.
9
+
10
+ ---
11
+
12
+ ## Features
13
+ 1. **Data Preprocessing**: Cleans and merges raw movie datasets.
14
+ 2. **Model Training**: Creates a similarity matrix using cosine similarity on vectorized features.
15
+ 3. **Fuzzy Matching**: Matches user input to the closest movie title using fuzzy string matching.
16
+ 4. **Recommendations**: Recommends movies with similarity scores.
17
+ 5. **Modular Design**: Separate modules for preprocessing, training, and recommendations.
18
+
19
+ ---
20
+
21
+ ## Project Structure
22
+ ```
23
+ ├── data
24
+ │ ├── credits.csv # Raw credits dataset
25
+ │ ├── movies.csv # Raw movies dataset
26
+ │ └── processed_dataset
27
+ │ └── movies_processed.csv # Preprocessed movie data
28
+ ├── src
29
+ │ ├── scripts
30
+ │ │ ├── preprocessing.py # Data preprocessing pipeline
31
+ │ │ ├── model.py # Model training and artifact creation
32
+ │ │ ├── recommender.py # Recommendation engine
33
+ │ └── trained_model
34
+ │ ├── similarity_matrix.pkl # Precomputed similarity matrix
35
+ │ ├── vectorizer.pkl # Saved CountVectorizer instance
36
+ │ └── feature_names.pkl # Feature names from vectorization
37
+ ├── main.py # Central script for running pipelines
38
+ ├── requirements.txt # Project dependencies
39
+ └── README.md # Project documentation
40
+ ```
41
+
42
+
43
+
44
+ ---
45
+
46
+ ## Technical Details
47
+ ### Preprocessing
48
+ - Extracts key features (`genres`, `keywords`, `cast`, `crew`).
49
+ - Combines features into a unified text field (`tags`) for vectorization.
50
+
51
+ ### Model
52
+ - **Vectorization**: Converts text into numerical vectors using `CountVectorizer`.
53
+ - **Similarity Matrix**: Computes cosine similarity between movie vectors.
54
+
55
+ ### Recommendation Logic
56
+ 1. Matches user input to the closest movie title using fuzzy string matching.
57
+ 2. Fetches the most similar movies based on the similarity matrix.
58
+ 3. Outputs a ranked list of recommendations with similarity scores.
59
+
60
+ ---
61
+
62
+ ## Example Output
63
+ ```bash
64
+ Recommendations for 'The Dark Knight':
65
+ - The Dark Knight Rises (95.67% match)
66
+ - Batman Begins (89.23% match)
67
+ - Inception (82.45% match)
68
+ ```
69
+
70
+ ---
71
+
72
+ ## Challenges and Solutions
73
+ - **Large Dataset**: Optimized vectorization using `CountVectorizer` with a feature limit.
74
+ - **Matching Errors**: Improved title matching accuracy with fuzzy string matching.
75
+
76
+ ---
77
+
78
+ ## Future Enhancements
79
+ 1. Implement collaborative filtering for better recommendations.
80
+ 2. Introduce a user-friendly web interface.
81
+ 3. Integrate additional data sources, such as IMDb ratings and reviews.
82
+
83
+ ---
84
+
85
+ ## Contributing
86
+ Contributions are welcome! Fork the repository, make changes, and submit a pull request.
87
+
88
+ ---
89
+
90
+ ## Contact
91
+ For questions or suggestions, feel free to reach out:
92
+ - Email: [maazuddin173@gmail.com](mailto:maazuddin173@gmail.com)
93
+ - GitHub: [Maazuddin1](https://github.com/Maazuddin1)
data/credits.csv ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2671838f8f178271eecf4e7e62463eb2ad85c40f2213067c614329eb7cb8ca0a
3
+ size 40049097
data/movies.csv ADDED
The diff for this file is too large to render. See raw diff
 
data/processed_dataset/movies_preprocessed.csv ADDED
The diff for this file is too large to render. See raw diff
 
notebook/movie_recommendation_system.ipynb ADDED
The diff for this file is too large to render. See raw diff
 
requirements.txt ADDED
Binary file (1.03 kB). View file
 
setup.py ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from setuptools import setup, find_packages
2
+
3
+ setup(
4
+ name="Content based Movie Recommedation system",
5
+ version="1.0",
6
+ description="A machine learning project for Movie Recommedation",
7
+ author="Maaz uddin",
8
+ packages=find_packages(),
9
+ install_requires=[
10
+ "flask",
11
+ "pandas",
12
+ "numpy",
13
+ "scikit-learn",
14
+ "seaborn",
15
+ "matplotlib"
16
+ ]
17
+ )
src/app/app.py ADDED
@@ -0,0 +1,29 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from flask import Flask, render_template, request
2
+ import sys
3
+ import os
4
+ from pathlib import Path
5
+ # Add project root to sys.path
6
+ project_root = Path(__file__).resolve().parents[2]
7
+ sys.path.append(str(project_root))
8
+ from src.scripts.recommender import MovieRecommender
9
+ # app.py (Flask backend)
10
+ app = Flask(__name__,template_folder='src/templates')
11
+
12
+ recommender = MovieRecommender(model_dir='src/trained_model')
13
+
14
+ @app.route('/', methods=['GET', 'POST'])
15
+ def home():
16
+ recommendations = []
17
+ matched_title = ""
18
+
19
+ if request.method == 'POST':
20
+ movie_title = request.form.get('movie_title')
21
+ recommender.load_model_artifacts()
22
+ recommendations, matched_title = recommender.recommend_movies(movie_title)
23
+
24
+ return render_template('index.html', recommendations=recommendations, matched_title=matched_title)
25
+
26
+ if __name__ == '__main__':
27
+ debug_mode = os.getenv('FLASK_DEBUG', 'False').lower() in ['true', '1', 'yes']
28
+ app.run(debug=debug_mode)
29
+
src/app/main.py ADDED
@@ -0,0 +1,80 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import sys
2
+ import logging
3
+ import sys
4
+ from pathlib import Path
5
+ # Add project root to sys.path
6
+ project_root = Path(__file__).resolve().parents[2]
7
+ sys.path.append(str(project_root))
8
+
9
+
10
+ # Import from scripts
11
+ from src.scripts.preprocessing import DataPreprocessor
12
+ from src.scripts.recommender import MovieRecommender
13
+
14
+ # Set up logging
15
+ logging.basicConfig(
16
+ level=logging.INFO,
17
+ format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
18
+ )
19
+ logger = logging.getLogger(__name__)
20
+
21
+ def main():
22
+ try:
23
+ # Define base project path
24
+ base_path = Path(__file__).resolve().parents[2]
25
+
26
+ # File paths
27
+ credits_path = base_path/'data'/'credits.csv'
28
+ movies_path = base_path/'data'/'movies.csv'
29
+ processed_path = base_path/'data'/'processed_dataset'/'movies_preprocessed.csv'
30
+ model_dir = base_path/'src'/'trained_model'
31
+
32
+ # Ensure necessary directories exist
33
+ model_dir.mkdir(parents=True, exist_ok=True)
34
+ processed_path.parent.mkdir(parents=True, exist_ok=True)
35
+
36
+
37
+ # Data Preprocessing
38
+ logger.info("Starting data preprocessing...")
39
+ preprocessor = DataPreprocessor(
40
+ credits_path=credits_path,
41
+ movies_path=movies_path,
42
+ output_path=processed_path
43
+ )
44
+ preprocessor.load_data()
45
+ processed_df = preprocessor.preprocess_data()
46
+ preprocessor.save_preprocessed_data()
47
+ logger.info("Data preprocessing completed successfully.")
48
+
49
+
50
+
51
+ # Model Training
52
+ logger.info("Starting model training...")
53
+ recommender = MovieRecommender(model_dir=str(model_dir))
54
+ recommender.create_similarity_matrix(processed_df)
55
+ logger.info("Model training completed successfully.")
56
+ # Model Testing with a Sample Recommendation
57
+ logger.info("Testing model with a sample recommendation...")
58
+ recommender.load_model_artifacts()
59
+ movie_title = "Dark Knight"
60
+ recommendations, matched_title = recommender.recommend_movies(movie_title)
61
+
62
+
63
+
64
+ # Display Recommendations
65
+ if recommendations and matched_title:
66
+ print(f"\nRecommendations for '{matched_title}':")
67
+ for movie in recommendations:
68
+ print(f"- {movie['title']} ({movie['similarity']}% match)")
69
+ else:
70
+ print(f"No recommendations found for '{movie_title}'.")
71
+
72
+ except Exception as e:
73
+ logger.error(f"Error in main: {str(e)}")
74
+ raise
75
+
76
+ if __name__ == "__main__":
77
+ # Add project root to sys.path for module resolution
78
+ project_root = Path(__file__).resolve().parents[2]
79
+ sys.path.append(str(project_root))
80
+ main()
src/scripts/EDA.py ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # src/scripts/EDA.py
2
+ import pandas as pd
3
+ import matplotlib.pyplot as plt
4
+ from pathlib import Path
5
+
6
+ def perform_eda(credits_df, movies_df):
7
+ """Perform exploratory data analysis on the datasets."""
8
+ print("Dataset Information:")
9
+ print("\nMovies Dataset Shape:", movies_df.shape)
10
+ print("Credits Dataset Shape:", credits_df.shape)
11
+
12
+
13
+ pd.options.display.float_format = '{:.2f}'.format
14
+
15
+
16
+ # Missing values analysis
17
+ print("\nMissing Values:")
18
+ print(movies_df.isnull().sum())
19
+
20
+ # Basic statistics
21
+ print("\nNumerical Columns Statistics:")
22
+ print(movies_df.describe())
23
+
24
+ # Generate visualizations
25
+ plt.figure(figsize=(10, 6))
26
+ movies_df['vote_average'].hist()
27
+ plt.title('Distribution of Movie Ratings')
28
+ plt.xlabel('Rating')
29
+ plt.ylabel('Count')
30
+ #plt.savefig('data/visualizations/ratings_dist.png')
31
+ #plt.close()
32
+
33
+
34
+ def main():
35
+ # Load the preprocessed data
36
+ credits_df = pd.read_csv('g:/my projects/Content-based-Movie_Recommedation_system/data/credits.csv')
37
+ movies_df = pd.read_csv('g:/my projects/Content-based-Movie_Recommedation_system/data/movies.csv')
38
+
39
+ # Perform EDA
40
+ perform_eda(credits_df, movies_df)
41
+
42
+ if __name__ == "__main__":
43
+ main()
src/scripts/model.py ADDED
@@ -0,0 +1,52 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # src/scripts/train_model.py
2
+ from pathlib import Path
3
+ from preprocessing import DataPreprocessor
4
+ import sys
5
+ import os
6
+ from recommender import MovieRecommender
7
+ import logging
8
+ import pandas as pd
9
+
10
+ # Set up logging
11
+ logging.basicConfig(
12
+ level=logging.INFO,
13
+ format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
14
+ )
15
+ logger = logging.getLogger(__name__)
16
+
17
+ def main():
18
+ try:
19
+ # Get absolute paths
20
+ base_path = Path(__file__).resolve().parents[2]
21
+
22
+ # Setup paths
23
+ credits_path = base_path/'data'/'credits.csv'
24
+ movies_path = base_path/'data'/'movies.csv'
25
+ processed_path = base_path/'data'/'processed_dataset'/'movies_preprocessed.csv'
26
+ model_dir = base_path/'src'/'trained_model'
27
+
28
+ logger.info("Preprocessed data found. Loading it directly...")
29
+ processed_df = pd.read_csv(processed_path) # Adjust based on file format (e.g., CSV, pickle)
30
+ logger.info("Starting model training...")
31
+
32
+ # Initialize and train recommender
33
+ recommender = MovieRecommender(model_dir=str(model_dir))
34
+ recommender.create_similarity_matrix(processed_df)
35
+ logger.info("Model training completed successfully!")
36
+
37
+ # Test the model
38
+ logger.info("Testing model with a sample recommendation...")
39
+ recommender.load_model_artifacts()
40
+ recommendations, matched_title = recommender.recommend_movies("The Dark Knight")
41
+
42
+ if recommendations and matched_title:
43
+ print(f"\nRecommendations for {matched_title}:")
44
+ for movie in recommendations:
45
+ print(f"- {movie['title']} ({movie['similarity']}% match)")
46
+
47
+ except Exception as e:
48
+ logger.error(f"Error in main: {str(e)}")
49
+ raise
50
+
51
+ if __name__ == "__main__":
52
+ main()
src/scripts/preprocessing.py ADDED
@@ -0,0 +1,131 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # src/scripts/preprocessing.py
2
+ import pandas as pd
3
+ import numpy as np
4
+ import ast
5
+ import logging
6
+ from pathlib import Path
7
+
8
+ logging.basicConfig(
9
+ level=logging.INFO,
10
+ format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
11
+ )
12
+ logger = logging.getLogger(__name__)
13
+
14
+ class DataPreprocessor:
15
+ def __init__(self, credits_path,movies_path,output_path):
16
+ self.credits_path = Path(credits_path)
17
+ self.movies_path = Path(movies_path)
18
+ self.output_path = Path(output_path)
19
+ self.data = None
20
+
21
+ def load_data(self):
22
+ """Load and merge the movies and credits datasets."""
23
+ try:
24
+ logger.info("Loading datasets...")
25
+ credits_df = pd.read_csv(self.credits_path)
26
+ movies_df = pd.read_csv(self.movies_path)
27
+ self.data = movies_df.merge(credits_df)
28
+ logger.info(f"Successfully loaded {len(self.data)} movies")
29
+ return self.data
30
+ except Exception as e:
31
+ logger.error(f"Error loading data: {str(e)}")
32
+ raise
33
+
34
+ @staticmethod
35
+ def convert_literals(obj):
36
+ """Convert string literals to list of names."""
37
+ return [item['name'] for item in ast.literal_eval(obj)]
38
+
39
+ @staticmethod
40
+ def cast3only(obj):
41
+ """Extract top 3 cast members."""
42
+ cast = []
43
+ counter = 0
44
+ for i in ast.literal_eval(obj):
45
+ if counter != 3:
46
+ cast.append(i['name'])
47
+ counter += 1
48
+ return cast
49
+
50
+ @staticmethod
51
+ def find_director(obj):
52
+ """Extract director name from crew."""
53
+ for i in ast.literal_eval(obj):
54
+ if i['job'] == 'Director':
55
+ return [i['name']]
56
+ return []
57
+
58
+ def preprocess_data(self):
59
+ """Main preprocessing function."""
60
+ try:
61
+ if self.data is None:
62
+ self.load_data()
63
+
64
+ logger.info("Preprocessing data...")
65
+ # Select required columns
66
+ self.data = self.data[['movie_id', 'title', 'overview', 'genres', 'keywords', 'cast', 'crew']]
67
+
68
+ # Drop missing values
69
+ self.data = self.data.dropna()
70
+
71
+ # Apply transformations
72
+ self.data['genres'] = self.data['genres'].apply(self.convert_literals)
73
+ self.data['keywords'] = self.data['keywords'].apply(self.convert_literals)
74
+ self.data['cast'] = self.data['cast'].apply(self.cast3only)
75
+ self.data['crew'] = self.data['crew'].apply(self.find_director)
76
+
77
+ # Process text data
78
+ self.data['overview'] = self.data['overview'].apply(lambda x: x.split())
79
+
80
+ # Remove spaces from all lists
81
+ for col in ['overview', 'genres', 'keywords', 'cast']:
82
+ self.data[col] = self.data[col].apply(lambda x: [i.replace(' ', '') for i in x])
83
+
84
+ # Combine features
85
+ self.data['tags'] = self.data['overview'] + self.data['genres'] + self.data['keywords'] + self.data['cast'] + self.data['crew']
86
+
87
+ self.data = self.data[['movie_id', 'title', 'tags']]
88
+ self.data['tags'] = self.data['tags'].apply(lambda x: ' '.join(x))
89
+ self.data['tags'] = self.data['tags'].apply(lambda x: x.lower())
90
+
91
+ logger.info("Data preprocessing completed.")
92
+ return self.data
93
+
94
+ except Exception as e:
95
+ logger.error(f"Error during preprocessing: {str(e)}")
96
+ raise
97
+
98
+ def save_preprocessed_data(self):
99
+ """Save preprocessed data to a csv file."""
100
+ try:
101
+ self.output_path.parent.mkdir(parents=True, exist_ok=True)
102
+ self.data.to_csv(self.output_path, index=False)
103
+ logger.info(f"Saved preprocessed data to {self.output_path}")
104
+ except Exception as e:
105
+ logger.error(f"Error saving preprocessed data: {str(e)}")
106
+ raise
107
+
108
+ def main():
109
+ # Set file paths
110
+ credits_path = 'data/credits.csv'
111
+ movies_path = 'data/movies.csv'
112
+ output_path = 'data/processed_dataset/movies_preprocessed.csv'
113
+
114
+ # Create an instance of DataPreprocessor
115
+ preprocessor = DataPreprocessor(credits_path, movies_path, output_path)
116
+
117
+ try:
118
+ # Load and preprocess data
119
+ preprocessor.load_data()
120
+ preprocessor.preprocess_data()
121
+ # Save preprocessed data
122
+ preprocessor.save_preprocessed_data()
123
+
124
+ logger.info("Preprocessing pipeline executed successfully.")
125
+
126
+ except Exception as e:
127
+ logger.error(f"An error occurred during the preprocessing pipeline: {str(e)}")
128
+
129
+
130
+ if __name__ == '__main__':
131
+ main()
src/scripts/recommender.py ADDED
@@ -0,0 +1,142 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # src/scripts/recommender.py
2
+
3
+ # THIS A FILE FUNCTIONS IS CALLED BY "model.py"
4
+ # THIS FILE CONTAINS FUNCTIONS REQUIRED BY model.py
5
+
6
+
7
+
8
+ import pandas as pd
9
+ from sklearn.feature_extraction.text import CountVectorizer
10
+ from sklearn.metrics.pairwise import cosine_similarity
11
+ from fuzzywuzzy import fuzz, process
12
+ import pickle
13
+ import logging
14
+ from pathlib import Path
15
+
16
+ # Set up logging
17
+ logging.basicConfig(
18
+ level=logging.INFO,
19
+ format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
20
+ )
21
+ logger = logging.getLogger(__name__)
22
+
23
+ class MovieRecommender:
24
+ def __init__(self, model_dir='src/trained_model'):
25
+ self.model_dir = Path(model_dir)
26
+ self.processed_data_path = self.model_dir/'processed_data.pkl'
27
+ self.similarity_matrix_path = self.model_dir/'similarity_matrix.pkl'
28
+ self.vectorizer_path = self.model_dir/'vectorizer.pkl'
29
+ self.feature_names_path = self.model_dir/'feature_names.pkl'
30
+ self.df = None
31
+ self.similarity_matrix = None
32
+ self.vectorizer = None
33
+
34
+ # to create and save trained model pickle files
35
+ def create_similarity_matrix(self,df):
36
+ """Create and save similarity matrix from processed data."""
37
+ try:
38
+ logger.info("Creating similarity matrix...")
39
+
40
+ # Create vectors
41
+ cv = CountVectorizer(max_features=5000, stop_words='english')
42
+ vectors = cv.fit_transform(df['tags']).toarray()
43
+ similarity_matrix = cosine_similarity(vectors)
44
+
45
+ # Create models directory if it doesn't exist
46
+ self.model_dir.mkdir(parents=True, exist_ok=True)
47
+
48
+ # Save all artifacts
49
+ logger.info("Saving model artifacts...")
50
+ with open(self.vectorizer_path, 'wb') as f:
51
+ pickle.dump(cv, f)
52
+
53
+ with open(self.similarity_matrix_path, 'wb') as f:
54
+ pickle.dump(similarity_matrix, f)
55
+
56
+ df.to_pickle(self.processed_data_path)
57
+
58
+ feature_names = cv.get_feature_names_out()
59
+ with open(self.feature_names_path, 'wb') as f:
60
+ pickle.dump(feature_names, f)
61
+
62
+ logger.info("Successfully created and saved all model artifacts")
63
+ self.df = df
64
+ self.similarity_matrix = similarity_matrix
65
+ self.vectorizer = cv
66
+
67
+ return similarity_matrix
68
+
69
+ except Exception as e:
70
+ logger.error(f"Error in create_similarity_matrix: {str(e)}")
71
+ raise
72
+
73
+ # CALLED BY "model.py" TO READ PICKLE FILES FOR PREDICTION
74
+ def load_model_artifacts(self):
75
+ """Load saved model artifacts."""
76
+ try:
77
+ logger.info("Loading model artifacts...")
78
+ self.df = pd.read_pickle(self.processed_data_path)
79
+ with open(self.similarity_matrix_path, 'rb') as f:
80
+ self.similarity_matrix = pickle.load(f)
81
+ with open(self.vectorizer_path, 'rb') as f:
82
+ self.vectorizer = pickle.load(f)
83
+ logger.info("Model artifacts loaded successfully.")
84
+ except Exception as e:
85
+ logger.error(f"Error loading model artifacts: {str(e)}")
86
+ raise
87
+
88
+
89
+
90
+ # this will be called in recommend_movies 👇method(below fuction)
91
+ def find_closest_title(self, input_title, score_cutoff=60):
92
+ """Find the closest matching movie title using fuzzy string matching."""
93
+ input_title = input_title.lower()
94
+ title_list = self.df['title'].tolist()
95
+ matches = process.extractBests(
96
+ input_title,
97
+ title_list,
98
+ scorer=fuzz.token_sort_ratio,
99
+ score_cutoff=score_cutoff,
100
+ limit=5
101
+ )
102
+ return matches[0][0] if matches else None
103
+ def recommend_movies(self, movie_title, n_recommendations=5):
104
+ """Get movie recommendations based on similarity with fuzzy matching."""
105
+ try:
106
+ if self.df is None or self.similarity_matrix is None:
107
+ raise ValueError("Model artifacts are not loaded. Please load them first.")
108
+
109
+ # Find closest matching title
110
+ matched_title = self.find_closest_title(movie_title)
111
+
112
+ if matched_title is None:
113
+ logger.warning(f"No close matches found for '{movie_title}'")
114
+ return [], None
115
+
116
+ # Get movie index
117
+ movie_index = self.df[self.df['title'] == matched_title].index[0]
118
+ distances = self.similarity_matrix[movie_index]
119
+
120
+ # Get similar movies
121
+ movie_list = sorted(list(enumerate(distances)),
122
+ reverse=True,
123
+ key=lambda x: x[1])[1:n_recommendations+1]
124
+
125
+ # Get recommendations with similarity scores
126
+ recommendations = [
127
+ {
128
+ 'title': self.df.iloc[i[0]].title,
129
+ 'similarity': round(i[1] * 100, 2)
130
+ }
131
+ for i in movie_list
132
+ ]
133
+
134
+ return recommendations, matched_title
135
+
136
+ except Exception as e:
137
+ logger.error(f"Error generating recommendations: {str(e)}")
138
+ return [], None
139
+
140
+
141
+ if __name__ == "__main__":
142
+ pass
src/templates/index.html ADDED
@@ -0,0 +1,162 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ <!DOCTYPE html>
2
+ <html lang="en">
3
+ <head>
4
+ <meta charset="UTF-8">
5
+ <meta name="viewport" content="width=device-width, initial-scale=1.0">
6
+ <title>Movie Recommendation System</title>
7
+ <style>
8
+ body {
9
+ font-family: Arial, sans-serif;
10
+ margin: 0;
11
+ padding: 0;
12
+ background-color: #f7f7f7;
13
+ color: #333;
14
+ }
15
+
16
+ header {
17
+ background-color: #3e8e41;
18
+ padding: 20px;
19
+ text-align: center;
20
+ color: white;
21
+ }
22
+
23
+ .container {
24
+ max-width: 900px;
25
+ margin: 0 auto;
26
+ padding: 20px;
27
+ }
28
+
29
+ h1 {
30
+ font-size: 36px;
31
+ margin-bottom: 20px;
32
+ }
33
+
34
+ .form-container {
35
+ background-color: white;
36
+ padding: 20px;
37
+ border-radius: 5px;
38
+ box-shadow: 0 4px 8px rgba(0, 0, 0, 0.1);
39
+ margin-bottom: 30px;
40
+ }
41
+
42
+ label {
43
+ font-size: 18px;
44
+ margin-bottom: 10px;
45
+ display: inline-block;
46
+ }
47
+
48
+ input[type="text"] {
49
+ width: 100%;
50
+ padding: 10px;
51
+ margin-top: 5px;
52
+ border: 1px solid #ddd;
53
+ border-radius: 5px;
54
+ font-size: 16px;
55
+ }
56
+
57
+ button {
58
+ padding: 10px 20px;
59
+ background-color: #3e8e41;
60
+ color: white;
61
+ border: none;
62
+ border-radius: 5px;
63
+ font-size: 16px;
64
+ cursor: pointer;
65
+ margin-top: 20px;
66
+ width: 100%;
67
+ }
68
+
69
+ button:hover {
70
+ background-color: #2d6a29;
71
+ }
72
+
73
+ .recommendations {
74
+ background-color: white;
75
+ padding: 20px;
76
+ border-radius: 5px;
77
+ box-shadow: 0 4px 8px rgba(0, 0, 0, 0.1);
78
+ }
79
+
80
+ .recommendations h2 {
81
+ font-size: 28px;
82
+ margin-bottom: 20px;
83
+ }
84
+
85
+ .recommendations ul {
86
+ list-style: none;
87
+ padding: 0;
88
+ }
89
+
90
+ .recommendations li {
91
+ background-color: #f9f9f9;
92
+ margin-bottom: 10px;
93
+ padding: 10px;
94
+ border-radius: 5px;
95
+ font-size: 18px;
96
+ display: flex;
97
+ justify-content: space-between;
98
+ align-items: center;
99
+ }
100
+
101
+ .recommendations li span {
102
+ font-size: 14px;
103
+ color: #666;
104
+ }
105
+
106
+ .error-message {
107
+ background-color: #ffcccb;
108
+ padding: 10px;
109
+ border-radius: 5px;
110
+ text-align: center;
111
+ color: #d8000c;
112
+ margin-bottom: 20px;
113
+ }
114
+ </style>
115
+ </head>
116
+ <body>
117
+
118
+ <header>
119
+ <h1>Movie Recommendation System</h1>
120
+ </header>
121
+
122
+ <div class="container">
123
+ <div class="form-container">
124
+ <h2>Get Movie Recommendations</h2>
125
+ <p>Enter a movie title below, and we will suggest similar movies for you!</p>
126
+
127
+ <form action="/" method="POST">
128
+ <label for="movie_title">Movie Title:</label>
129
+ <input type="text" id="movie_title" name="movie_title" required placeholder="e.g., The Dark Knight">
130
+ <button type="submit">Get Recommendations</button>
131
+ </form>
132
+ </div>
133
+
134
+ <!-- Display Error if no recommendations are found -->
135
+ {% if not recommendations %}
136
+ <div class="error-message">
137
+ <strong>No recommendations found. Please try a different movie title.</strong>
138
+ </div>
139
+ {% endif %}
140
+
141
+ <!-- Display Recommendations -->
142
+ {% if recommendations %}
143
+ <div class="recommendations">
144
+ <h2>Recommendations for "{{ matched_title }}"</h2>
145
+ <ul>
146
+ {% for movie in recommendations %}
147
+ <li>
148
+ <div>
149
+ <strong>{{ movie['title'] }}</strong>
150
+ </div>
151
+ <div>
152
+ <span>{{ movie['similarity'] }}% match</span>
153
+ </div>
154
+ </li>
155
+ {% endfor %}
156
+ </ul>
157
+ </div>
158
+ {% endif %}
159
+ </div>
160
+
161
+ </body>
162
+ </html>
src/trained_model/feature_names.pkl ADDED
Binary file (51.7 kB). View file
 
src/trained_model/processed_data.pkl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:62114f41cbb99ba6360b1d199877b3ddbfd42bd96d012b7b4a1d3e10d537a610
3
+ size 2396007
src/trained_model/similarity_matrix.pkl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c9e97fd615845dcc92e67c58ea7c066ed419a3b8fecb769a81e6f7a7b219f0d0
3
+ size 184704363
src/trained_model/vectorizer.pkl ADDED
Binary file (516 kB). View file
 
tests/test_model.py ADDED
@@ -0,0 +1,57 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import pytest
2
+ from src.scripts.recommender import MovieRecommender
3
+ import pandas as pd
4
+ import numpy as np
5
+
6
+ @pytest.fixture
7
+ def sample_movies_df():
8
+ return pd.DataFrame({
9
+ 'movie_id': [1, 2, 3],
10
+ 'title': ['The Dark Knight', 'Inception', 'Interstellar'],
11
+ 'tags': [
12
+ 'batman dark knight action',
13
+ 'dreams inception thriller',
14
+ 'space interstellar scifi'
15
+ ]
16
+ })
17
+
18
+ def test_movie_recommender_initialization():
19
+ recommender = MovieRecommender()
20
+ assert recommender is not None
21
+ assert recommender.df is None
22
+ assert recommender.similarity_matrix is None
23
+
24
+ def test_find_closest_title(sample_movies_df):
25
+ recommender = MovieRecommender()
26
+ recommender.df = sample_movies_df
27
+
28
+ # Test exact match
29
+ assert recommender.find_closest_title('The Dark Knight') == 'The Dark Knight'
30
+
31
+ # Test partial match
32
+ assert recommender.find_closest_title('Dark Knight') == 'The Dark Knight'
33
+
34
+ # Test no match
35
+ assert recommender.find_closest_title('Nonexistent Movie') is None
36
+
37
+ def test_recommend_movies(sample_movies_df):
38
+ recommender = MovieRecommender()
39
+ recommender.df = sample_movies_df
40
+
41
+ # Create a simple similarity matrix for testing
42
+ recommender.similarity_matrix = np.array([
43
+ [1.0, 0.5, 0.3],
44
+ [0.5, 1.0, 0.4],
45
+ [0.3, 0.4, 1.0]
46
+ ])
47
+
48
+ # Test recommendations
49
+ recommendations, matched_title = recommender.recommend_movies('The Dark Knight')
50
+
51
+ assert matched_title == 'The Dark Knight'
52
+ assert len(recommendations) == 2 # Should return 2 recommendations
53
+ assert recommendations[0]['title'] in ['Inception', 'Interstellar']
54
+ assert 0 <= recommendations[0]['similarity'] <= 100
55
+
56
+ if __name__ == '__main__':
57
+ pytest.main(['-v'])