Maaz commited on
Upload 22 files
Browse files- .gitattributes +4 -35
- .github/workflows/python-app.yml +87 -0
- .gitignore +171 -0
- README.md +93 -10
- data/credits.csv +3 -0
- data/movies.csv +0 -0
- data/processed_dataset/movies_preprocessed.csv +0 -0
- notebook/movie_recommendation_system.ipynb +0 -0
- requirements.txt +0 -0
- setup.py +17 -0
- src/app/app.py +29 -0
- src/app/main.py +80 -0
- src/scripts/EDA.py +43 -0
- src/scripts/model.py +52 -0
- src/scripts/preprocessing.py +131 -0
- src/scripts/recommender.py +142 -0
- src/templates/index.html +162 -0
- src/trained_model/feature_names.pkl +0 -0
- src/trained_model/processed_data.pkl +3 -0
- src/trained_model/similarity_matrix.pkl +3 -0
- src/trained_model/vectorizer.pkl +0 -0
- tests/test_model.py +57 -0
.gitattributes
CHANGED
|
@@ -1,35 +1,4 @@
|
|
| 1 |
-
|
| 2 |
-
*.
|
| 3 |
-
|
| 4 |
-
|
| 5 |
-
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
| 6 |
-
*.ftz filter=lfs diff=lfs merge=lfs -text
|
| 7 |
-
*.gz filter=lfs diff=lfs merge=lfs -text
|
| 8 |
-
*.h5 filter=lfs diff=lfs merge=lfs -text
|
| 9 |
-
*.joblib filter=lfs diff=lfs merge=lfs -text
|
| 10 |
-
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
| 11 |
-
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
| 12 |
-
*.model filter=lfs diff=lfs merge=lfs -text
|
| 13 |
-
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
| 14 |
-
*.npy filter=lfs diff=lfs merge=lfs -text
|
| 15 |
-
*.npz filter=lfs diff=lfs merge=lfs -text
|
| 16 |
-
*.onnx filter=lfs diff=lfs merge=lfs -text
|
| 17 |
-
*.ot filter=lfs diff=lfs merge=lfs -text
|
| 18 |
-
*.parquet filter=lfs diff=lfs merge=lfs -text
|
| 19 |
-
*.pb filter=lfs diff=lfs merge=lfs -text
|
| 20 |
-
*.pickle filter=lfs diff=lfs merge=lfs -text
|
| 21 |
-
*.pkl filter=lfs diff=lfs merge=lfs -text
|
| 22 |
-
*.pt filter=lfs diff=lfs merge=lfs -text
|
| 23 |
-
*.pth filter=lfs diff=lfs merge=lfs -text
|
| 24 |
-
*.rar filter=lfs diff=lfs merge=lfs -text
|
| 25 |
-
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 26 |
-
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
| 27 |
-
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
| 28 |
-
*.tar filter=lfs diff=lfs merge=lfs -text
|
| 29 |
-
*.tflite filter=lfs diff=lfs merge=lfs -text
|
| 30 |
-
*.tgz filter=lfs diff=lfs merge=lfs -text
|
| 31 |
-
*.wasm filter=lfs diff=lfs merge=lfs -text
|
| 32 |
-
*.xz filter=lfs diff=lfs merge=lfs -text
|
| 33 |
-
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
-
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
-
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
| 1 |
+
src/trained_model/similarity_matrix.pkl filter=lfs diff=lfs merge=lfs -text
|
| 2 |
+
*.ipynb linguist-detectable=false
|
| 3 |
+
data/credits.csv filter=lfs diff=lfs merge=lfs -text
|
| 4 |
+
src/trained_model/processed_data.pkl filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
.github/workflows/python-app.yml
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# This workflow will install Python dependencies, run tests and lint with a single version of Python
|
| 2 |
+
# For more information see: https://docs.github.com/en/actions/automating-builds-and-tests/building-and-testing-python
|
| 3 |
+
|
| 4 |
+
name: Python application
|
| 5 |
+
|
| 6 |
+
on:
|
| 7 |
+
push:
|
| 8 |
+
branches: [ "main" ]
|
| 9 |
+
pull_request:
|
| 10 |
+
branches: [ "main" ]
|
| 11 |
+
|
| 12 |
+
permissions:
|
| 13 |
+
contents: read
|
| 14 |
+
|
| 15 |
+
jobs:
|
| 16 |
+
build:
|
| 17 |
+
runs-on: ubuntu-latest
|
| 18 |
+
steps:
|
| 19 |
+
- uses: actions/checkout@v4
|
| 20 |
+
|
| 21 |
+
- name: Set up Python 3.10
|
| 22 |
+
uses: actions/setup-python@v3
|
| 23 |
+
with:
|
| 24 |
+
python-version: "3.10"
|
| 25 |
+
|
| 26 |
+
- name: Install dependencies
|
| 27 |
+
run: |
|
| 28 |
+
python -m pip install --upgrade pip setuptools wheel
|
| 29 |
+
pip install flake8 pytest pytest-flask pandas scikit-learn flask fuzzywuzzy python-Levenshtein
|
| 30 |
+
if [ -f requirements.txt ]; then pip install -r requirements.txt; fi
|
| 31 |
+
|
| 32 |
+
- name: Verify data and model directories
|
| 33 |
+
run: |
|
| 34 |
+
mkdir -p data/processed_dataset
|
| 35 |
+
mkdir -p src/trained_model
|
| 36 |
+
touch data/movies.csv data/credits.csv
|
| 37 |
+
if [ ! -d "src/trained_model" ]; then
|
| 38 |
+
echo "Model directory missing!"
|
| 39 |
+
exit 1
|
| 40 |
+
fi
|
| 41 |
+
if [ ! -d "src/templates" ] || [ ! -f "src/templates/index.html" ]; then
|
| 42 |
+
echo "Template files missing!"
|
| 43 |
+
exit 1
|
| 44 |
+
fi
|
| 45 |
+
|
| 46 |
+
- name: Lint with flake8
|
| 47 |
+
run: |
|
| 48 |
+
flake8 . --count --select=E9,F63,F7,F82 --show-source --statistics
|
| 49 |
+
flake8 . --count --exit-zero --max-complexity=10 --max-line-length=127 --statistics
|
| 50 |
+
|
| 51 |
+
- name: Process data and train model
|
| 52 |
+
run: |
|
| 53 |
+
python src/app/main.py
|
| 54 |
+
|
| 55 |
+
- name: Verify model artifacts
|
| 56 |
+
run: |
|
| 57 |
+
if [ ! -f "src/trained_model/processed_data.pkl" ] || \
|
| 58 |
+
[ ! -f "src/trained_model/similarity_matrix.pkl" ] || \
|
| 59 |
+
[ ! -f "src/trained_model/vectorizer.pkl" ]; then
|
| 60 |
+
echo "Model artifacts missing!"
|
| 61 |
+
exit 1
|
| 62 |
+
fi
|
| 63 |
+
|
| 64 |
+
- name: Test with pytest
|
| 65 |
+
run: |
|
| 66 |
+
python -m pytest tests/ -v
|
| 67 |
+
|
| 68 |
+
- name: Start and Test Flask App
|
| 69 |
+
run: |
|
| 70 |
+
python src/app/app.py &
|
| 71 |
+
sleep 10
|
| 72 |
+
curl --retry 5 --retry-delay 5 --retry-connrefused http://127.0.0.1:5000/ || exit 1
|
| 73 |
+
pkill -f "python src/app/app.py"
|
| 74 |
+
env:
|
| 75 |
+
FLASK_ENV: testing
|
| 76 |
+
FLASK_DEBUG: 0
|
| 77 |
+
|
| 78 |
+
|
| 79 |
+
- name: Check setup.py
|
| 80 |
+
run: |
|
| 81 |
+
if [ -f setup.py ]; then
|
| 82 |
+
python setup.py check
|
| 83 |
+
python setup.py sdist bdist_wheel
|
| 84 |
+
pip install -e .
|
| 85 |
+
else
|
| 86 |
+
echo "setup.py not found - skipping package build"
|
| 87 |
+
fi
|
.gitignore
ADDED
|
@@ -0,0 +1,171 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Byte-compiled / optimized / DLL files
|
| 2 |
+
__pycache__/
|
| 3 |
+
*.py[cod]
|
| 4 |
+
*$py.class
|
| 5 |
+
|
| 6 |
+
# C extensions
|
| 7 |
+
*.so
|
| 8 |
+
|
| 9 |
+
# Distribution / packaging
|
| 10 |
+
.Python
|
| 11 |
+
build/
|
| 12 |
+
develop-eggs/
|
| 13 |
+
dist/
|
| 14 |
+
downloads/
|
| 15 |
+
eggs/
|
| 16 |
+
.eggs/
|
| 17 |
+
lib/
|
| 18 |
+
lib64/
|
| 19 |
+
parts/
|
| 20 |
+
sdist/
|
| 21 |
+
var/
|
| 22 |
+
wheels/
|
| 23 |
+
share/python-wheels/
|
| 24 |
+
*.egg-info/
|
| 25 |
+
.installed.cfg
|
| 26 |
+
*.egg
|
| 27 |
+
MANIFEST
|
| 28 |
+
|
| 29 |
+
# PyInstaller
|
| 30 |
+
# Usually these files are written by a python script from a template
|
| 31 |
+
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
| 32 |
+
*.manifest
|
| 33 |
+
*.spec
|
| 34 |
+
|
| 35 |
+
# Installer logs
|
| 36 |
+
pip-log.txt
|
| 37 |
+
pip-delete-this-directory.txt
|
| 38 |
+
|
| 39 |
+
# Unit test / coverage reports
|
| 40 |
+
htmlcov/
|
| 41 |
+
.tox/
|
| 42 |
+
.nox/
|
| 43 |
+
.coverage
|
| 44 |
+
.coverage.*
|
| 45 |
+
.cache
|
| 46 |
+
nosetests.xml
|
| 47 |
+
coverage.xml
|
| 48 |
+
*.cover
|
| 49 |
+
*.py,cover
|
| 50 |
+
.hypothesis/
|
| 51 |
+
.pytest_cache/
|
| 52 |
+
cover/
|
| 53 |
+
|
| 54 |
+
# Translations
|
| 55 |
+
*.mo
|
| 56 |
+
*.pot
|
| 57 |
+
|
| 58 |
+
# Django stuff:
|
| 59 |
+
*.log
|
| 60 |
+
local_settings.py
|
| 61 |
+
db.sqlite3
|
| 62 |
+
db.sqlite3-journal
|
| 63 |
+
|
| 64 |
+
# Flask stuff:
|
| 65 |
+
instance/
|
| 66 |
+
.webassets-cache
|
| 67 |
+
|
| 68 |
+
# Scrapy stuff:
|
| 69 |
+
.scrapy
|
| 70 |
+
|
| 71 |
+
# Sphinx documentation
|
| 72 |
+
docs/_build/
|
| 73 |
+
|
| 74 |
+
# PyBuilder
|
| 75 |
+
.pybuilder/
|
| 76 |
+
target/
|
| 77 |
+
|
| 78 |
+
# Jupyter Notebook
|
| 79 |
+
.ipynb_checkpoints
|
| 80 |
+
|
| 81 |
+
# IPython
|
| 82 |
+
profile_default/
|
| 83 |
+
ipython_config.py
|
| 84 |
+
|
| 85 |
+
# pyenv
|
| 86 |
+
# For a library or package, you might want to ignore these files since the code is
|
| 87 |
+
# intended to run in multiple environments; otherwise, check them in:
|
| 88 |
+
# .python-version
|
| 89 |
+
|
| 90 |
+
# pipenv
|
| 91 |
+
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
|
| 92 |
+
# However, in case of collaboration, if having platform-specific dependencies or dependencies
|
| 93 |
+
# having no cross-platform support, pipenv may install dependencies that don't work, or not
|
| 94 |
+
# install all needed dependencies.
|
| 95 |
+
#Pipfile.lock
|
| 96 |
+
|
| 97 |
+
# UV
|
| 98 |
+
# Similar to Pipfile.lock, it is generally recommended to include uv.lock in version control.
|
| 99 |
+
# This is especially recommended for binary packages to ensure reproducibility, and is more
|
| 100 |
+
# commonly ignored for libraries.
|
| 101 |
+
#uv.lock
|
| 102 |
+
|
| 103 |
+
# poetry
|
| 104 |
+
# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
|
| 105 |
+
# This is especially recommended for binary packages to ensure reproducibility, and is more
|
| 106 |
+
# commonly ignored for libraries.
|
| 107 |
+
# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
|
| 108 |
+
#poetry.lock
|
| 109 |
+
|
| 110 |
+
# pdm
|
| 111 |
+
# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
|
| 112 |
+
#pdm.lock
|
| 113 |
+
# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
|
| 114 |
+
# in version control.
|
| 115 |
+
# https://pdm.fming.dev/latest/usage/project/#working-with-version-control
|
| 116 |
+
.pdm.toml
|
| 117 |
+
.pdm-python
|
| 118 |
+
.pdm-build/
|
| 119 |
+
|
| 120 |
+
# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
|
| 121 |
+
__pypackages__/
|
| 122 |
+
|
| 123 |
+
# Celery stuff
|
| 124 |
+
celerybeat-schedule
|
| 125 |
+
celerybeat.pid
|
| 126 |
+
|
| 127 |
+
# SageMath parsed files
|
| 128 |
+
*.sage.py
|
| 129 |
+
|
| 130 |
+
# Environments
|
| 131 |
+
.env
|
| 132 |
+
.venv
|
| 133 |
+
env/
|
| 134 |
+
venv/
|
| 135 |
+
ENV/
|
| 136 |
+
env.bak/
|
| 137 |
+
venv.bak/
|
| 138 |
+
|
| 139 |
+
# Spyder project settings
|
| 140 |
+
.spyderproject
|
| 141 |
+
.spyproject
|
| 142 |
+
|
| 143 |
+
# Rope project settings
|
| 144 |
+
.ropeproject
|
| 145 |
+
|
| 146 |
+
# mkdocs documentation
|
| 147 |
+
/site
|
| 148 |
+
|
| 149 |
+
# mypy
|
| 150 |
+
.mypy_cache/
|
| 151 |
+
.dmypy.json
|
| 152 |
+
dmypy.json
|
| 153 |
+
|
| 154 |
+
# Pyre type checker
|
| 155 |
+
.pyre/
|
| 156 |
+
|
| 157 |
+
# pytype static type analyzer
|
| 158 |
+
.pytype/
|
| 159 |
+
|
| 160 |
+
# Cython debug symbols
|
| 161 |
+
cython_debug/
|
| 162 |
+
|
| 163 |
+
# PyCharm
|
| 164 |
+
# JetBrains specific template is maintained in a separate JetBrains.gitignore that can
|
| 165 |
+
# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
|
| 166 |
+
# and can be added to the global gitignore or merged into this file. For a more nuclear
|
| 167 |
+
# option (not recommended) you can uncomment the following to ignore the entire idea folder.
|
| 168 |
+
#.idea/
|
| 169 |
+
|
| 170 |
+
# PyPI configuration file
|
| 171 |
+
.pypirc
|
README.md
CHANGED
|
@@ -1,10 +1,93 @@
|
|
| 1 |
-
|
| 2 |
-
|
| 3 |
-
|
| 4 |
-
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
| 8 |
-
-
|
| 9 |
-
|
| 10 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Movie Recommendation System
|
| 2 |
+
|
| 3 |
+
## 🎥 Watch the Demo👇
|
| 4 |
+
|
| 5 |
+

|
| 6 |
+
|
| 7 |
+
## Overview
|
| 8 |
+
The **Movie Recommendation System** is a Python-based project designed to recommend movies to users based on content similarity. Using natural language processing and machine learning techniques, this system analyzes movie metadata to provide tailored recommendations.
|
| 9 |
+
|
| 10 |
+
---
|
| 11 |
+
|
| 12 |
+
## Features
|
| 13 |
+
1. **Data Preprocessing**: Cleans and merges raw movie datasets.
|
| 14 |
+
2. **Model Training**: Creates a similarity matrix using cosine similarity on vectorized features.
|
| 15 |
+
3. **Fuzzy Matching**: Matches user input to the closest movie title using fuzzy string matching.
|
| 16 |
+
4. **Recommendations**: Recommends movies with similarity scores.
|
| 17 |
+
5. **Modular Design**: Separate modules for preprocessing, training, and recommendations.
|
| 18 |
+
|
| 19 |
+
---
|
| 20 |
+
|
| 21 |
+
## Project Structure
|
| 22 |
+
```
|
| 23 |
+
├── data
|
| 24 |
+
│ ├── credits.csv # Raw credits dataset
|
| 25 |
+
│ ├── movies.csv # Raw movies dataset
|
| 26 |
+
│ └── processed_dataset
|
| 27 |
+
│ └── movies_processed.csv # Preprocessed movie data
|
| 28 |
+
├── src
|
| 29 |
+
│ ├── scripts
|
| 30 |
+
│ │ ├── preprocessing.py # Data preprocessing pipeline
|
| 31 |
+
│ │ ├── model.py # Model training and artifact creation
|
| 32 |
+
│ │ ├── recommender.py # Recommendation engine
|
| 33 |
+
│ └── trained_model
|
| 34 |
+
│ ├── similarity_matrix.pkl # Precomputed similarity matrix
|
| 35 |
+
│ ├── vectorizer.pkl # Saved CountVectorizer instance
|
| 36 |
+
│ └── feature_names.pkl # Feature names from vectorization
|
| 37 |
+
├── main.py # Central script for running pipelines
|
| 38 |
+
├── requirements.txt # Project dependencies
|
| 39 |
+
└── README.md # Project documentation
|
| 40 |
+
```
|
| 41 |
+
|
| 42 |
+
|
| 43 |
+
|
| 44 |
+
---
|
| 45 |
+
|
| 46 |
+
## Technical Details
|
| 47 |
+
### Preprocessing
|
| 48 |
+
- Extracts key features (`genres`, `keywords`, `cast`, `crew`).
|
| 49 |
+
- Combines features into a unified text field (`tags`) for vectorization.
|
| 50 |
+
|
| 51 |
+
### Model
|
| 52 |
+
- **Vectorization**: Converts text into numerical vectors using `CountVectorizer`.
|
| 53 |
+
- **Similarity Matrix**: Computes cosine similarity between movie vectors.
|
| 54 |
+
|
| 55 |
+
### Recommendation Logic
|
| 56 |
+
1. Matches user input to the closest movie title using fuzzy string matching.
|
| 57 |
+
2. Fetches the most similar movies based on the similarity matrix.
|
| 58 |
+
3. Outputs a ranked list of recommendations with similarity scores.
|
| 59 |
+
|
| 60 |
+
---
|
| 61 |
+
|
| 62 |
+
## Example Output
|
| 63 |
+
```bash
|
| 64 |
+
Recommendations for 'The Dark Knight':
|
| 65 |
+
- The Dark Knight Rises (95.67% match)
|
| 66 |
+
- Batman Begins (89.23% match)
|
| 67 |
+
- Inception (82.45% match)
|
| 68 |
+
```
|
| 69 |
+
|
| 70 |
+
---
|
| 71 |
+
|
| 72 |
+
## Challenges and Solutions
|
| 73 |
+
- **Large Dataset**: Optimized vectorization using `CountVectorizer` with a feature limit.
|
| 74 |
+
- **Matching Errors**: Improved title matching accuracy with fuzzy string matching.
|
| 75 |
+
|
| 76 |
+
---
|
| 77 |
+
|
| 78 |
+
## Future Enhancements
|
| 79 |
+
1. Implement collaborative filtering for better recommendations.
|
| 80 |
+
2. Introduce a user-friendly web interface.
|
| 81 |
+
3. Integrate additional data sources, such as IMDb ratings and reviews.
|
| 82 |
+
|
| 83 |
+
---
|
| 84 |
+
|
| 85 |
+
## Contributing
|
| 86 |
+
Contributions are welcome! Fork the repository, make changes, and submit a pull request.
|
| 87 |
+
|
| 88 |
+
---
|
| 89 |
+
|
| 90 |
+
## Contact
|
| 91 |
+
For questions or suggestions, feel free to reach out:
|
| 92 |
+
- Email: [maazuddin173@gmail.com](mailto:maazuddin173@gmail.com)
|
| 93 |
+
- GitHub: [Maazuddin1](https://github.com/Maazuddin1)
|
data/credits.csv
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2671838f8f178271eecf4e7e62463eb2ad85c40f2213067c614329eb7cb8ca0a
|
| 3 |
+
size 40049097
|
data/movies.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/processed_dataset/movies_preprocessed.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
notebook/movie_recommendation_system.ipynb
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
requirements.txt
ADDED
|
Binary file (1.03 kB). View file
|
|
|
setup.py
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from setuptools import setup, find_packages
|
| 2 |
+
|
| 3 |
+
setup(
|
| 4 |
+
name="Content based Movie Recommedation system",
|
| 5 |
+
version="1.0",
|
| 6 |
+
description="A machine learning project for Movie Recommedation",
|
| 7 |
+
author="Maaz uddin",
|
| 8 |
+
packages=find_packages(),
|
| 9 |
+
install_requires=[
|
| 10 |
+
"flask",
|
| 11 |
+
"pandas",
|
| 12 |
+
"numpy",
|
| 13 |
+
"scikit-learn",
|
| 14 |
+
"seaborn",
|
| 15 |
+
"matplotlib"
|
| 16 |
+
]
|
| 17 |
+
)
|
src/app/app.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from flask import Flask, render_template, request
|
| 2 |
+
import sys
|
| 3 |
+
import os
|
| 4 |
+
from pathlib import Path
|
| 5 |
+
# Add project root to sys.path
|
| 6 |
+
project_root = Path(__file__).resolve().parents[2]
|
| 7 |
+
sys.path.append(str(project_root))
|
| 8 |
+
from src.scripts.recommender import MovieRecommender
|
| 9 |
+
# app.py (Flask backend)
|
| 10 |
+
app = Flask(__name__,template_folder='src/templates')
|
| 11 |
+
|
| 12 |
+
recommender = MovieRecommender(model_dir='src/trained_model')
|
| 13 |
+
|
| 14 |
+
@app.route('/', methods=['GET', 'POST'])
|
| 15 |
+
def home():
|
| 16 |
+
recommendations = []
|
| 17 |
+
matched_title = ""
|
| 18 |
+
|
| 19 |
+
if request.method == 'POST':
|
| 20 |
+
movie_title = request.form.get('movie_title')
|
| 21 |
+
recommender.load_model_artifacts()
|
| 22 |
+
recommendations, matched_title = recommender.recommend_movies(movie_title)
|
| 23 |
+
|
| 24 |
+
return render_template('index.html', recommendations=recommendations, matched_title=matched_title)
|
| 25 |
+
|
| 26 |
+
if __name__ == '__main__':
|
| 27 |
+
debug_mode = os.getenv('FLASK_DEBUG', 'False').lower() in ['true', '1', 'yes']
|
| 28 |
+
app.run(debug=debug_mode)
|
| 29 |
+
|
src/app/main.py
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import sys
|
| 2 |
+
import logging
|
| 3 |
+
import sys
|
| 4 |
+
from pathlib import Path
|
| 5 |
+
# Add project root to sys.path
|
| 6 |
+
project_root = Path(__file__).resolve().parents[2]
|
| 7 |
+
sys.path.append(str(project_root))
|
| 8 |
+
|
| 9 |
+
|
| 10 |
+
# Import from scripts
|
| 11 |
+
from src.scripts.preprocessing import DataPreprocessor
|
| 12 |
+
from src.scripts.recommender import MovieRecommender
|
| 13 |
+
|
| 14 |
+
# Set up logging
|
| 15 |
+
logging.basicConfig(
|
| 16 |
+
level=logging.INFO,
|
| 17 |
+
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
|
| 18 |
+
)
|
| 19 |
+
logger = logging.getLogger(__name__)
|
| 20 |
+
|
| 21 |
+
def main():
|
| 22 |
+
try:
|
| 23 |
+
# Define base project path
|
| 24 |
+
base_path = Path(__file__).resolve().parents[2]
|
| 25 |
+
|
| 26 |
+
# File paths
|
| 27 |
+
credits_path = base_path/'data'/'credits.csv'
|
| 28 |
+
movies_path = base_path/'data'/'movies.csv'
|
| 29 |
+
processed_path = base_path/'data'/'processed_dataset'/'movies_preprocessed.csv'
|
| 30 |
+
model_dir = base_path/'src'/'trained_model'
|
| 31 |
+
|
| 32 |
+
# Ensure necessary directories exist
|
| 33 |
+
model_dir.mkdir(parents=True, exist_ok=True)
|
| 34 |
+
processed_path.parent.mkdir(parents=True, exist_ok=True)
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
# Data Preprocessing
|
| 38 |
+
logger.info("Starting data preprocessing...")
|
| 39 |
+
preprocessor = DataPreprocessor(
|
| 40 |
+
credits_path=credits_path,
|
| 41 |
+
movies_path=movies_path,
|
| 42 |
+
output_path=processed_path
|
| 43 |
+
)
|
| 44 |
+
preprocessor.load_data()
|
| 45 |
+
processed_df = preprocessor.preprocess_data()
|
| 46 |
+
preprocessor.save_preprocessed_data()
|
| 47 |
+
logger.info("Data preprocessing completed successfully.")
|
| 48 |
+
|
| 49 |
+
|
| 50 |
+
|
| 51 |
+
# Model Training
|
| 52 |
+
logger.info("Starting model training...")
|
| 53 |
+
recommender = MovieRecommender(model_dir=str(model_dir))
|
| 54 |
+
recommender.create_similarity_matrix(processed_df)
|
| 55 |
+
logger.info("Model training completed successfully.")
|
| 56 |
+
# Model Testing with a Sample Recommendation
|
| 57 |
+
logger.info("Testing model with a sample recommendation...")
|
| 58 |
+
recommender.load_model_artifacts()
|
| 59 |
+
movie_title = "Dark Knight"
|
| 60 |
+
recommendations, matched_title = recommender.recommend_movies(movie_title)
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
# Display Recommendations
|
| 65 |
+
if recommendations and matched_title:
|
| 66 |
+
print(f"\nRecommendations for '{matched_title}':")
|
| 67 |
+
for movie in recommendations:
|
| 68 |
+
print(f"- {movie['title']} ({movie['similarity']}% match)")
|
| 69 |
+
else:
|
| 70 |
+
print(f"No recommendations found for '{movie_title}'.")
|
| 71 |
+
|
| 72 |
+
except Exception as e:
|
| 73 |
+
logger.error(f"Error in main: {str(e)}")
|
| 74 |
+
raise
|
| 75 |
+
|
| 76 |
+
if __name__ == "__main__":
|
| 77 |
+
# Add project root to sys.path for module resolution
|
| 78 |
+
project_root = Path(__file__).resolve().parents[2]
|
| 79 |
+
sys.path.append(str(project_root))
|
| 80 |
+
main()
|
src/scripts/EDA.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# src/scripts/EDA.py
|
| 2 |
+
import pandas as pd
|
| 3 |
+
import matplotlib.pyplot as plt
|
| 4 |
+
from pathlib import Path
|
| 5 |
+
|
| 6 |
+
def perform_eda(credits_df, movies_df):
|
| 7 |
+
"""Perform exploratory data analysis on the datasets."""
|
| 8 |
+
print("Dataset Information:")
|
| 9 |
+
print("\nMovies Dataset Shape:", movies_df.shape)
|
| 10 |
+
print("Credits Dataset Shape:", credits_df.shape)
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
pd.options.display.float_format = '{:.2f}'.format
|
| 14 |
+
|
| 15 |
+
|
| 16 |
+
# Missing values analysis
|
| 17 |
+
print("\nMissing Values:")
|
| 18 |
+
print(movies_df.isnull().sum())
|
| 19 |
+
|
| 20 |
+
# Basic statistics
|
| 21 |
+
print("\nNumerical Columns Statistics:")
|
| 22 |
+
print(movies_df.describe())
|
| 23 |
+
|
| 24 |
+
# Generate visualizations
|
| 25 |
+
plt.figure(figsize=(10, 6))
|
| 26 |
+
movies_df['vote_average'].hist()
|
| 27 |
+
plt.title('Distribution of Movie Ratings')
|
| 28 |
+
plt.xlabel('Rating')
|
| 29 |
+
plt.ylabel('Count')
|
| 30 |
+
#plt.savefig('data/visualizations/ratings_dist.png')
|
| 31 |
+
#plt.close()
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
def main():
|
| 35 |
+
# Load the preprocessed data
|
| 36 |
+
credits_df = pd.read_csv('g:/my projects/Content-based-Movie_Recommedation_system/data/credits.csv')
|
| 37 |
+
movies_df = pd.read_csv('g:/my projects/Content-based-Movie_Recommedation_system/data/movies.csv')
|
| 38 |
+
|
| 39 |
+
# Perform EDA
|
| 40 |
+
perform_eda(credits_df, movies_df)
|
| 41 |
+
|
| 42 |
+
if __name__ == "__main__":
|
| 43 |
+
main()
|
src/scripts/model.py
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# src/scripts/train_model.py
|
| 2 |
+
from pathlib import Path
|
| 3 |
+
from preprocessing import DataPreprocessor
|
| 4 |
+
import sys
|
| 5 |
+
import os
|
| 6 |
+
from recommender import MovieRecommender
|
| 7 |
+
import logging
|
| 8 |
+
import pandas as pd
|
| 9 |
+
|
| 10 |
+
# Set up logging
|
| 11 |
+
logging.basicConfig(
|
| 12 |
+
level=logging.INFO,
|
| 13 |
+
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
|
| 14 |
+
)
|
| 15 |
+
logger = logging.getLogger(__name__)
|
| 16 |
+
|
| 17 |
+
def main():
|
| 18 |
+
try:
|
| 19 |
+
# Get absolute paths
|
| 20 |
+
base_path = Path(__file__).resolve().parents[2]
|
| 21 |
+
|
| 22 |
+
# Setup paths
|
| 23 |
+
credits_path = base_path/'data'/'credits.csv'
|
| 24 |
+
movies_path = base_path/'data'/'movies.csv'
|
| 25 |
+
processed_path = base_path/'data'/'processed_dataset'/'movies_preprocessed.csv'
|
| 26 |
+
model_dir = base_path/'src'/'trained_model'
|
| 27 |
+
|
| 28 |
+
logger.info("Preprocessed data found. Loading it directly...")
|
| 29 |
+
processed_df = pd.read_csv(processed_path) # Adjust based on file format (e.g., CSV, pickle)
|
| 30 |
+
logger.info("Starting model training...")
|
| 31 |
+
|
| 32 |
+
# Initialize and train recommender
|
| 33 |
+
recommender = MovieRecommender(model_dir=str(model_dir))
|
| 34 |
+
recommender.create_similarity_matrix(processed_df)
|
| 35 |
+
logger.info("Model training completed successfully!")
|
| 36 |
+
|
| 37 |
+
# Test the model
|
| 38 |
+
logger.info("Testing model with a sample recommendation...")
|
| 39 |
+
recommender.load_model_artifacts()
|
| 40 |
+
recommendations, matched_title = recommender.recommend_movies("The Dark Knight")
|
| 41 |
+
|
| 42 |
+
if recommendations and matched_title:
|
| 43 |
+
print(f"\nRecommendations for {matched_title}:")
|
| 44 |
+
for movie in recommendations:
|
| 45 |
+
print(f"- {movie['title']} ({movie['similarity']}% match)")
|
| 46 |
+
|
| 47 |
+
except Exception as e:
|
| 48 |
+
logger.error(f"Error in main: {str(e)}")
|
| 49 |
+
raise
|
| 50 |
+
|
| 51 |
+
if __name__ == "__main__":
|
| 52 |
+
main()
|
src/scripts/preprocessing.py
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# src/scripts/preprocessing.py
|
| 2 |
+
import pandas as pd
|
| 3 |
+
import numpy as np
|
| 4 |
+
import ast
|
| 5 |
+
import logging
|
| 6 |
+
from pathlib import Path
|
| 7 |
+
|
| 8 |
+
logging.basicConfig(
|
| 9 |
+
level=logging.INFO,
|
| 10 |
+
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
|
| 11 |
+
)
|
| 12 |
+
logger = logging.getLogger(__name__)
|
| 13 |
+
|
| 14 |
+
class DataPreprocessor:
|
| 15 |
+
def __init__(self, credits_path,movies_path,output_path):
|
| 16 |
+
self.credits_path = Path(credits_path)
|
| 17 |
+
self.movies_path = Path(movies_path)
|
| 18 |
+
self.output_path = Path(output_path)
|
| 19 |
+
self.data = None
|
| 20 |
+
|
| 21 |
+
def load_data(self):
|
| 22 |
+
"""Load and merge the movies and credits datasets."""
|
| 23 |
+
try:
|
| 24 |
+
logger.info("Loading datasets...")
|
| 25 |
+
credits_df = pd.read_csv(self.credits_path)
|
| 26 |
+
movies_df = pd.read_csv(self.movies_path)
|
| 27 |
+
self.data = movies_df.merge(credits_df)
|
| 28 |
+
logger.info(f"Successfully loaded {len(self.data)} movies")
|
| 29 |
+
return self.data
|
| 30 |
+
except Exception as e:
|
| 31 |
+
logger.error(f"Error loading data: {str(e)}")
|
| 32 |
+
raise
|
| 33 |
+
|
| 34 |
+
@staticmethod
|
| 35 |
+
def convert_literals(obj):
|
| 36 |
+
"""Convert string literals to list of names."""
|
| 37 |
+
return [item['name'] for item in ast.literal_eval(obj)]
|
| 38 |
+
|
| 39 |
+
@staticmethod
|
| 40 |
+
def cast3only(obj):
|
| 41 |
+
"""Extract top 3 cast members."""
|
| 42 |
+
cast = []
|
| 43 |
+
counter = 0
|
| 44 |
+
for i in ast.literal_eval(obj):
|
| 45 |
+
if counter != 3:
|
| 46 |
+
cast.append(i['name'])
|
| 47 |
+
counter += 1
|
| 48 |
+
return cast
|
| 49 |
+
|
| 50 |
+
@staticmethod
|
| 51 |
+
def find_director(obj):
|
| 52 |
+
"""Extract director name from crew."""
|
| 53 |
+
for i in ast.literal_eval(obj):
|
| 54 |
+
if i['job'] == 'Director':
|
| 55 |
+
return [i['name']]
|
| 56 |
+
return []
|
| 57 |
+
|
| 58 |
+
def preprocess_data(self):
|
| 59 |
+
"""Main preprocessing function."""
|
| 60 |
+
try:
|
| 61 |
+
if self.data is None:
|
| 62 |
+
self.load_data()
|
| 63 |
+
|
| 64 |
+
logger.info("Preprocessing data...")
|
| 65 |
+
# Select required columns
|
| 66 |
+
self.data = self.data[['movie_id', 'title', 'overview', 'genres', 'keywords', 'cast', 'crew']]
|
| 67 |
+
|
| 68 |
+
# Drop missing values
|
| 69 |
+
self.data = self.data.dropna()
|
| 70 |
+
|
| 71 |
+
# Apply transformations
|
| 72 |
+
self.data['genres'] = self.data['genres'].apply(self.convert_literals)
|
| 73 |
+
self.data['keywords'] = self.data['keywords'].apply(self.convert_literals)
|
| 74 |
+
self.data['cast'] = self.data['cast'].apply(self.cast3only)
|
| 75 |
+
self.data['crew'] = self.data['crew'].apply(self.find_director)
|
| 76 |
+
|
| 77 |
+
# Process text data
|
| 78 |
+
self.data['overview'] = self.data['overview'].apply(lambda x: x.split())
|
| 79 |
+
|
| 80 |
+
# Remove spaces from all lists
|
| 81 |
+
for col in ['overview', 'genres', 'keywords', 'cast']:
|
| 82 |
+
self.data[col] = self.data[col].apply(lambda x: [i.replace(' ', '') for i in x])
|
| 83 |
+
|
| 84 |
+
# Combine features
|
| 85 |
+
self.data['tags'] = self.data['overview'] + self.data['genres'] + self.data['keywords'] + self.data['cast'] + self.data['crew']
|
| 86 |
+
|
| 87 |
+
self.data = self.data[['movie_id', 'title', 'tags']]
|
| 88 |
+
self.data['tags'] = self.data['tags'].apply(lambda x: ' '.join(x))
|
| 89 |
+
self.data['tags'] = self.data['tags'].apply(lambda x: x.lower())
|
| 90 |
+
|
| 91 |
+
logger.info("Data preprocessing completed.")
|
| 92 |
+
return self.data
|
| 93 |
+
|
| 94 |
+
except Exception as e:
|
| 95 |
+
logger.error(f"Error during preprocessing: {str(e)}")
|
| 96 |
+
raise
|
| 97 |
+
|
| 98 |
+
def save_preprocessed_data(self):
|
| 99 |
+
"""Save preprocessed data to a csv file."""
|
| 100 |
+
try:
|
| 101 |
+
self.output_path.parent.mkdir(parents=True, exist_ok=True)
|
| 102 |
+
self.data.to_csv(self.output_path, index=False)
|
| 103 |
+
logger.info(f"Saved preprocessed data to {self.output_path}")
|
| 104 |
+
except Exception as e:
|
| 105 |
+
logger.error(f"Error saving preprocessed data: {str(e)}")
|
| 106 |
+
raise
|
| 107 |
+
|
| 108 |
+
def main():
|
| 109 |
+
# Set file paths
|
| 110 |
+
credits_path = 'data/credits.csv'
|
| 111 |
+
movies_path = 'data/movies.csv'
|
| 112 |
+
output_path = 'data/processed_dataset/movies_preprocessed.csv'
|
| 113 |
+
|
| 114 |
+
# Create an instance of DataPreprocessor
|
| 115 |
+
preprocessor = DataPreprocessor(credits_path, movies_path, output_path)
|
| 116 |
+
|
| 117 |
+
try:
|
| 118 |
+
# Load and preprocess data
|
| 119 |
+
preprocessor.load_data()
|
| 120 |
+
preprocessor.preprocess_data()
|
| 121 |
+
# Save preprocessed data
|
| 122 |
+
preprocessor.save_preprocessed_data()
|
| 123 |
+
|
| 124 |
+
logger.info("Preprocessing pipeline executed successfully.")
|
| 125 |
+
|
| 126 |
+
except Exception as e:
|
| 127 |
+
logger.error(f"An error occurred during the preprocessing pipeline: {str(e)}")
|
| 128 |
+
|
| 129 |
+
|
| 130 |
+
if __name__ == '__main__':
|
| 131 |
+
main()
|
src/scripts/recommender.py
ADDED
|
@@ -0,0 +1,142 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# src/scripts/recommender.py
|
| 2 |
+
|
| 3 |
+
# THIS A FILE FUNCTIONS IS CALLED BY "model.py"
|
| 4 |
+
# THIS FILE CONTAINS FUNCTIONS REQUIRED BY model.py
|
| 5 |
+
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
import pandas as pd
|
| 9 |
+
from sklearn.feature_extraction.text import CountVectorizer
|
| 10 |
+
from sklearn.metrics.pairwise import cosine_similarity
|
| 11 |
+
from fuzzywuzzy import fuzz, process
|
| 12 |
+
import pickle
|
| 13 |
+
import logging
|
| 14 |
+
from pathlib import Path
|
| 15 |
+
|
| 16 |
+
# Set up logging
|
| 17 |
+
logging.basicConfig(
|
| 18 |
+
level=logging.INFO,
|
| 19 |
+
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
|
| 20 |
+
)
|
| 21 |
+
logger = logging.getLogger(__name__)
|
| 22 |
+
|
| 23 |
+
class MovieRecommender:
|
| 24 |
+
def __init__(self, model_dir='src/trained_model'):
|
| 25 |
+
self.model_dir = Path(model_dir)
|
| 26 |
+
self.processed_data_path = self.model_dir/'processed_data.pkl'
|
| 27 |
+
self.similarity_matrix_path = self.model_dir/'similarity_matrix.pkl'
|
| 28 |
+
self.vectorizer_path = self.model_dir/'vectorizer.pkl'
|
| 29 |
+
self.feature_names_path = self.model_dir/'feature_names.pkl'
|
| 30 |
+
self.df = None
|
| 31 |
+
self.similarity_matrix = None
|
| 32 |
+
self.vectorizer = None
|
| 33 |
+
|
| 34 |
+
# to create and save trained model pickle files
|
| 35 |
+
def create_similarity_matrix(self,df):
|
| 36 |
+
"""Create and save similarity matrix from processed data."""
|
| 37 |
+
try:
|
| 38 |
+
logger.info("Creating similarity matrix...")
|
| 39 |
+
|
| 40 |
+
# Create vectors
|
| 41 |
+
cv = CountVectorizer(max_features=5000, stop_words='english')
|
| 42 |
+
vectors = cv.fit_transform(df['tags']).toarray()
|
| 43 |
+
similarity_matrix = cosine_similarity(vectors)
|
| 44 |
+
|
| 45 |
+
# Create models directory if it doesn't exist
|
| 46 |
+
self.model_dir.mkdir(parents=True, exist_ok=True)
|
| 47 |
+
|
| 48 |
+
# Save all artifacts
|
| 49 |
+
logger.info("Saving model artifacts...")
|
| 50 |
+
with open(self.vectorizer_path, 'wb') as f:
|
| 51 |
+
pickle.dump(cv, f)
|
| 52 |
+
|
| 53 |
+
with open(self.similarity_matrix_path, 'wb') as f:
|
| 54 |
+
pickle.dump(similarity_matrix, f)
|
| 55 |
+
|
| 56 |
+
df.to_pickle(self.processed_data_path)
|
| 57 |
+
|
| 58 |
+
feature_names = cv.get_feature_names_out()
|
| 59 |
+
with open(self.feature_names_path, 'wb') as f:
|
| 60 |
+
pickle.dump(feature_names, f)
|
| 61 |
+
|
| 62 |
+
logger.info("Successfully created and saved all model artifacts")
|
| 63 |
+
self.df = df
|
| 64 |
+
self.similarity_matrix = similarity_matrix
|
| 65 |
+
self.vectorizer = cv
|
| 66 |
+
|
| 67 |
+
return similarity_matrix
|
| 68 |
+
|
| 69 |
+
except Exception as e:
|
| 70 |
+
logger.error(f"Error in create_similarity_matrix: {str(e)}")
|
| 71 |
+
raise
|
| 72 |
+
|
| 73 |
+
# CALLED BY "model.py" TO READ PICKLE FILES FOR PREDICTION
|
| 74 |
+
def load_model_artifacts(self):
|
| 75 |
+
"""Load saved model artifacts."""
|
| 76 |
+
try:
|
| 77 |
+
logger.info("Loading model artifacts...")
|
| 78 |
+
self.df = pd.read_pickle(self.processed_data_path)
|
| 79 |
+
with open(self.similarity_matrix_path, 'rb') as f:
|
| 80 |
+
self.similarity_matrix = pickle.load(f)
|
| 81 |
+
with open(self.vectorizer_path, 'rb') as f:
|
| 82 |
+
self.vectorizer = pickle.load(f)
|
| 83 |
+
logger.info("Model artifacts loaded successfully.")
|
| 84 |
+
except Exception as e:
|
| 85 |
+
logger.error(f"Error loading model artifacts: {str(e)}")
|
| 86 |
+
raise
|
| 87 |
+
|
| 88 |
+
|
| 89 |
+
|
| 90 |
+
# this will be called in recommend_movies 👇method(below fuction)
|
| 91 |
+
def find_closest_title(self, input_title, score_cutoff=60):
|
| 92 |
+
"""Find the closest matching movie title using fuzzy string matching."""
|
| 93 |
+
input_title = input_title.lower()
|
| 94 |
+
title_list = self.df['title'].tolist()
|
| 95 |
+
matches = process.extractBests(
|
| 96 |
+
input_title,
|
| 97 |
+
title_list,
|
| 98 |
+
scorer=fuzz.token_sort_ratio,
|
| 99 |
+
score_cutoff=score_cutoff,
|
| 100 |
+
limit=5
|
| 101 |
+
)
|
| 102 |
+
return matches[0][0] if matches else None
|
| 103 |
+
def recommend_movies(self, movie_title, n_recommendations=5):
|
| 104 |
+
"""Get movie recommendations based on similarity with fuzzy matching."""
|
| 105 |
+
try:
|
| 106 |
+
if self.df is None or self.similarity_matrix is None:
|
| 107 |
+
raise ValueError("Model artifacts are not loaded. Please load them first.")
|
| 108 |
+
|
| 109 |
+
# Find closest matching title
|
| 110 |
+
matched_title = self.find_closest_title(movie_title)
|
| 111 |
+
|
| 112 |
+
if matched_title is None:
|
| 113 |
+
logger.warning(f"No close matches found for '{movie_title}'")
|
| 114 |
+
return [], None
|
| 115 |
+
|
| 116 |
+
# Get movie index
|
| 117 |
+
movie_index = self.df[self.df['title'] == matched_title].index[0]
|
| 118 |
+
distances = self.similarity_matrix[movie_index]
|
| 119 |
+
|
| 120 |
+
# Get similar movies
|
| 121 |
+
movie_list = sorted(list(enumerate(distances)),
|
| 122 |
+
reverse=True,
|
| 123 |
+
key=lambda x: x[1])[1:n_recommendations+1]
|
| 124 |
+
|
| 125 |
+
# Get recommendations with similarity scores
|
| 126 |
+
recommendations = [
|
| 127 |
+
{
|
| 128 |
+
'title': self.df.iloc[i[0]].title,
|
| 129 |
+
'similarity': round(i[1] * 100, 2)
|
| 130 |
+
}
|
| 131 |
+
for i in movie_list
|
| 132 |
+
]
|
| 133 |
+
|
| 134 |
+
return recommendations, matched_title
|
| 135 |
+
|
| 136 |
+
except Exception as e:
|
| 137 |
+
logger.error(f"Error generating recommendations: {str(e)}")
|
| 138 |
+
return [], None
|
| 139 |
+
|
| 140 |
+
|
| 141 |
+
if __name__ == "__main__":
|
| 142 |
+
pass
|
src/templates/index.html
ADDED
|
@@ -0,0 +1,162 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
<!DOCTYPE html>
|
| 2 |
+
<html lang="en">
|
| 3 |
+
<head>
|
| 4 |
+
<meta charset="UTF-8">
|
| 5 |
+
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
| 6 |
+
<title>Movie Recommendation System</title>
|
| 7 |
+
<style>
|
| 8 |
+
body {
|
| 9 |
+
font-family: Arial, sans-serif;
|
| 10 |
+
margin: 0;
|
| 11 |
+
padding: 0;
|
| 12 |
+
background-color: #f7f7f7;
|
| 13 |
+
color: #333;
|
| 14 |
+
}
|
| 15 |
+
|
| 16 |
+
header {
|
| 17 |
+
background-color: #3e8e41;
|
| 18 |
+
padding: 20px;
|
| 19 |
+
text-align: center;
|
| 20 |
+
color: white;
|
| 21 |
+
}
|
| 22 |
+
|
| 23 |
+
.container {
|
| 24 |
+
max-width: 900px;
|
| 25 |
+
margin: 0 auto;
|
| 26 |
+
padding: 20px;
|
| 27 |
+
}
|
| 28 |
+
|
| 29 |
+
h1 {
|
| 30 |
+
font-size: 36px;
|
| 31 |
+
margin-bottom: 20px;
|
| 32 |
+
}
|
| 33 |
+
|
| 34 |
+
.form-container {
|
| 35 |
+
background-color: white;
|
| 36 |
+
padding: 20px;
|
| 37 |
+
border-radius: 5px;
|
| 38 |
+
box-shadow: 0 4px 8px rgba(0, 0, 0, 0.1);
|
| 39 |
+
margin-bottom: 30px;
|
| 40 |
+
}
|
| 41 |
+
|
| 42 |
+
label {
|
| 43 |
+
font-size: 18px;
|
| 44 |
+
margin-bottom: 10px;
|
| 45 |
+
display: inline-block;
|
| 46 |
+
}
|
| 47 |
+
|
| 48 |
+
input[type="text"] {
|
| 49 |
+
width: 100%;
|
| 50 |
+
padding: 10px;
|
| 51 |
+
margin-top: 5px;
|
| 52 |
+
border: 1px solid #ddd;
|
| 53 |
+
border-radius: 5px;
|
| 54 |
+
font-size: 16px;
|
| 55 |
+
}
|
| 56 |
+
|
| 57 |
+
button {
|
| 58 |
+
padding: 10px 20px;
|
| 59 |
+
background-color: #3e8e41;
|
| 60 |
+
color: white;
|
| 61 |
+
border: none;
|
| 62 |
+
border-radius: 5px;
|
| 63 |
+
font-size: 16px;
|
| 64 |
+
cursor: pointer;
|
| 65 |
+
margin-top: 20px;
|
| 66 |
+
width: 100%;
|
| 67 |
+
}
|
| 68 |
+
|
| 69 |
+
button:hover {
|
| 70 |
+
background-color: #2d6a29;
|
| 71 |
+
}
|
| 72 |
+
|
| 73 |
+
.recommendations {
|
| 74 |
+
background-color: white;
|
| 75 |
+
padding: 20px;
|
| 76 |
+
border-radius: 5px;
|
| 77 |
+
box-shadow: 0 4px 8px rgba(0, 0, 0, 0.1);
|
| 78 |
+
}
|
| 79 |
+
|
| 80 |
+
.recommendations h2 {
|
| 81 |
+
font-size: 28px;
|
| 82 |
+
margin-bottom: 20px;
|
| 83 |
+
}
|
| 84 |
+
|
| 85 |
+
.recommendations ul {
|
| 86 |
+
list-style: none;
|
| 87 |
+
padding: 0;
|
| 88 |
+
}
|
| 89 |
+
|
| 90 |
+
.recommendations li {
|
| 91 |
+
background-color: #f9f9f9;
|
| 92 |
+
margin-bottom: 10px;
|
| 93 |
+
padding: 10px;
|
| 94 |
+
border-radius: 5px;
|
| 95 |
+
font-size: 18px;
|
| 96 |
+
display: flex;
|
| 97 |
+
justify-content: space-between;
|
| 98 |
+
align-items: center;
|
| 99 |
+
}
|
| 100 |
+
|
| 101 |
+
.recommendations li span {
|
| 102 |
+
font-size: 14px;
|
| 103 |
+
color: #666;
|
| 104 |
+
}
|
| 105 |
+
|
| 106 |
+
.error-message {
|
| 107 |
+
background-color: #ffcccb;
|
| 108 |
+
padding: 10px;
|
| 109 |
+
border-radius: 5px;
|
| 110 |
+
text-align: center;
|
| 111 |
+
color: #d8000c;
|
| 112 |
+
margin-bottom: 20px;
|
| 113 |
+
}
|
| 114 |
+
</style>
|
| 115 |
+
</head>
|
| 116 |
+
<body>
|
| 117 |
+
|
| 118 |
+
<header>
|
| 119 |
+
<h1>Movie Recommendation System</h1>
|
| 120 |
+
</header>
|
| 121 |
+
|
| 122 |
+
<div class="container">
|
| 123 |
+
<div class="form-container">
|
| 124 |
+
<h2>Get Movie Recommendations</h2>
|
| 125 |
+
<p>Enter a movie title below, and we will suggest similar movies for you!</p>
|
| 126 |
+
|
| 127 |
+
<form action="/" method="POST">
|
| 128 |
+
<label for="movie_title">Movie Title:</label>
|
| 129 |
+
<input type="text" id="movie_title" name="movie_title" required placeholder="e.g., The Dark Knight">
|
| 130 |
+
<button type="submit">Get Recommendations</button>
|
| 131 |
+
</form>
|
| 132 |
+
</div>
|
| 133 |
+
|
| 134 |
+
<!-- Display Error if no recommendations are found -->
|
| 135 |
+
{% if not recommendations %}
|
| 136 |
+
<div class="error-message">
|
| 137 |
+
<strong>No recommendations found. Please try a different movie title.</strong>
|
| 138 |
+
</div>
|
| 139 |
+
{% endif %}
|
| 140 |
+
|
| 141 |
+
<!-- Display Recommendations -->
|
| 142 |
+
{% if recommendations %}
|
| 143 |
+
<div class="recommendations">
|
| 144 |
+
<h2>Recommendations for "{{ matched_title }}"</h2>
|
| 145 |
+
<ul>
|
| 146 |
+
{% for movie in recommendations %}
|
| 147 |
+
<li>
|
| 148 |
+
<div>
|
| 149 |
+
<strong>{{ movie['title'] }}</strong>
|
| 150 |
+
</div>
|
| 151 |
+
<div>
|
| 152 |
+
<span>{{ movie['similarity'] }}% match</span>
|
| 153 |
+
</div>
|
| 154 |
+
</li>
|
| 155 |
+
{% endfor %}
|
| 156 |
+
</ul>
|
| 157 |
+
</div>
|
| 158 |
+
{% endif %}
|
| 159 |
+
</div>
|
| 160 |
+
|
| 161 |
+
</body>
|
| 162 |
+
</html>
|
src/trained_model/feature_names.pkl
ADDED
|
Binary file (51.7 kB). View file
|
|
|
src/trained_model/processed_data.pkl
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:62114f41cbb99ba6360b1d199877b3ddbfd42bd96d012b7b4a1d3e10d537a610
|
| 3 |
+
size 2396007
|
src/trained_model/similarity_matrix.pkl
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c9e97fd615845dcc92e67c58ea7c066ed419a3b8fecb769a81e6f7a7b219f0d0
|
| 3 |
+
size 184704363
|
src/trained_model/vectorizer.pkl
ADDED
|
Binary file (516 kB). View file
|
|
|
tests/test_model.py
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import pytest
|
| 2 |
+
from src.scripts.recommender import MovieRecommender
|
| 3 |
+
import pandas as pd
|
| 4 |
+
import numpy as np
|
| 5 |
+
|
| 6 |
+
@pytest.fixture
|
| 7 |
+
def sample_movies_df():
|
| 8 |
+
return pd.DataFrame({
|
| 9 |
+
'movie_id': [1, 2, 3],
|
| 10 |
+
'title': ['The Dark Knight', 'Inception', 'Interstellar'],
|
| 11 |
+
'tags': [
|
| 12 |
+
'batman dark knight action',
|
| 13 |
+
'dreams inception thriller',
|
| 14 |
+
'space interstellar scifi'
|
| 15 |
+
]
|
| 16 |
+
})
|
| 17 |
+
|
| 18 |
+
def test_movie_recommender_initialization():
|
| 19 |
+
recommender = MovieRecommender()
|
| 20 |
+
assert recommender is not None
|
| 21 |
+
assert recommender.df is None
|
| 22 |
+
assert recommender.similarity_matrix is None
|
| 23 |
+
|
| 24 |
+
def test_find_closest_title(sample_movies_df):
|
| 25 |
+
recommender = MovieRecommender()
|
| 26 |
+
recommender.df = sample_movies_df
|
| 27 |
+
|
| 28 |
+
# Test exact match
|
| 29 |
+
assert recommender.find_closest_title('The Dark Knight') == 'The Dark Knight'
|
| 30 |
+
|
| 31 |
+
# Test partial match
|
| 32 |
+
assert recommender.find_closest_title('Dark Knight') == 'The Dark Knight'
|
| 33 |
+
|
| 34 |
+
# Test no match
|
| 35 |
+
assert recommender.find_closest_title('Nonexistent Movie') is None
|
| 36 |
+
|
| 37 |
+
def test_recommend_movies(sample_movies_df):
|
| 38 |
+
recommender = MovieRecommender()
|
| 39 |
+
recommender.df = sample_movies_df
|
| 40 |
+
|
| 41 |
+
# Create a simple similarity matrix for testing
|
| 42 |
+
recommender.similarity_matrix = np.array([
|
| 43 |
+
[1.0, 0.5, 0.3],
|
| 44 |
+
[0.5, 1.0, 0.4],
|
| 45 |
+
[0.3, 0.4, 1.0]
|
| 46 |
+
])
|
| 47 |
+
|
| 48 |
+
# Test recommendations
|
| 49 |
+
recommendations, matched_title = recommender.recommend_movies('The Dark Knight')
|
| 50 |
+
|
| 51 |
+
assert matched_title == 'The Dark Knight'
|
| 52 |
+
assert len(recommendations) == 2 # Should return 2 recommendations
|
| 53 |
+
assert recommendations[0]['title'] in ['Inception', 'Interstellar']
|
| 54 |
+
assert 0 <= recommendations[0]['similarity'] <= 100
|
| 55 |
+
|
| 56 |
+
if __name__ == '__main__':
|
| 57 |
+
pytest.main(['-v'])
|