Spaces:
Running on Zero
Running on Zero
Upload folder using huggingface_hub
Browse files- .idea/.gitignore +3 -0
- .idea/WhatsNew-Scraper.iml +9 -0
- .idea/caches/deviceStreaming.xml +0 -0
- .idea/misc.xml +19 -0
- .idea/modules.xml +8 -0
- .idea/studiobot.xml +6 -0
- .idea/vcs.xml +6 -0
- .idea/workspace.xml +55 -0
- __pycache__/main.cpython-314.pyc +0 -0
- app.py +269 -0
- data/__pycache__/firestore_repo.cpython-314.pyc +0 -0
- data/repositories/__pycache__/article_repository_impl.cpython-314.pyc +0 -0
- data/repositories/article_repository_impl.py +58 -0
- domain/__pycache__/interfaces.cpython-314.pyc +0 -0
- domain/__pycache__/models.cpython-314.pyc +0 -0
- domain/entities/__pycache__/article.cpython-314.pyc +0 -0
- domain/entities/article.py +74 -0
- domain/interfaces.py +10 -0
- domain/repositories/__pycache__/article_repository.cpython-314.pyc +0 -0
- domain/repositories/article_repository.py +14 -0
- domain/usecases/__pycache__/scrape_and_enrich.cpython-314.pyc +0 -0
- domain/usecases/scrape_and_enrich.py +69 -0
- main.py +133 -0
- nlp_training/__init__.py +1 -0
- nlp_training/prepare_datasets.py +398 -0
- nlp_training/train_category_classifier.py +190 -0
- nlp_training/train_ner_model.py +271 -0
- nlp_training/train_sentiment_analyzer.py +195 -0
- requirements.txt +54 -0
- requirements_space.txt +9 -0
- scrapers/__pycache__/rss_scraper.cpython-314.pyc +0 -0
- scrapers/base_scraper.py +21 -0
- scrapers/rss_scraper.py +128 -0
- services/__pycache__/ai_classifier.cpython-314.pyc +0 -0
- services/__pycache__/duplicate_detector.cpython-314.pyc +0 -0
- services/ai_classifier.py +185 -0
- services/duplicate_detector.py +45 -0
- services/lexicons/__init__.py +15 -0
- services/lexicons/__pycache__/__init__.cpython-314.pyc +0 -0
- services/lexicons/__pycache__/categories_lexicon.cpython-314.pyc +0 -0
- services/lexicons/__pycache__/images_lexicon.cpython-314.pyc +0 -0
- services/lexicons/__pycache__/locations_lexicon.cpython-314.pyc +0 -0
- services/lexicons/__pycache__/sentiment_lexicon.cpython-314.pyc +0 -0
- services/lexicons/__pycache__/stopwords_lexicon.cpython-314.pyc +0 -0
- services/lexicons/categories_lexicon.py +135 -0
- services/lexicons/images_lexicon.py +21 -0
- services/lexicons/locations_lexicon.py +59 -0
- services/lexicons/sentiment_lexicon.py +46 -0
- services/lexicons/stopwords_lexicon.py +20 -0
- services/local_nlp_classifier.py +399 -0
.idea/.gitignore
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Default ignored files
|
| 2 |
+
/shelf/
|
| 3 |
+
/workspace.xml
|
.idea/WhatsNew-Scraper.iml
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
<?xml version="1.0" encoding="UTF-8"?>
|
| 2 |
+
<module type="JAVA_MODULE" version="4">
|
| 3 |
+
<component name="NewModuleRootManager" inherit-compiler-output="true">
|
| 4 |
+
<exclude-output />
|
| 5 |
+
<content url="file://$MODULE_DIR$" />
|
| 6 |
+
<orderEntry type="inheritedJdk" />
|
| 7 |
+
<orderEntry type="sourceFolder" forTests="false" />
|
| 8 |
+
</component>
|
| 9 |
+
</module>
|
.idea/caches/deviceStreaming.xml
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
.idea/misc.xml
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
<?xml version="1.0" encoding="UTF-8"?>
|
| 2 |
+
<project version="4">
|
| 3 |
+
<component name="ProjectRootManager" version="2" languageLevel="JDK_21" default="true" project-jdk-name="jbr-21" project-jdk-type="JavaSDK">
|
| 4 |
+
<output url="file://$PROJECT_DIR$/out" />
|
| 5 |
+
</component>
|
| 6 |
+
<component name="StoredSessionsInfoStore">
|
| 7 |
+
<option name="sessions">
|
| 8 |
+
<list>
|
| 9 |
+
<SessionInfoState>
|
| 10 |
+
<option name="id" value="20260805-000758-9a7987ea-2159-4416-abc2-15f38fb7cc6d" />
|
| 11 |
+
<option name="lastUpdateTimestamp" value="1785877678080" />
|
| 12 |
+
<option name="responseMode" value="Agent" />
|
| 13 |
+
<option name="title" value="New Agent Session" />
|
| 14 |
+
<option name="user" value="abdo1234567es@gmail.com" />
|
| 15 |
+
</SessionInfoState>
|
| 16 |
+
</list>
|
| 17 |
+
</option>
|
| 18 |
+
</component>
|
| 19 |
+
</project>
|
.idea/modules.xml
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
<?xml version="1.0" encoding="UTF-8"?>
|
| 2 |
+
<project version="4">
|
| 3 |
+
<component name="ProjectModuleManager">
|
| 4 |
+
<modules>
|
| 5 |
+
<module fileurl="file://$PROJECT_DIR$/.idea/WhatsNew-Scraper.iml" filepath="$PROJECT_DIR$/.idea/WhatsNew-Scraper.iml" />
|
| 6 |
+
</modules>
|
| 7 |
+
</component>
|
| 8 |
+
</project>
|
.idea/studiobot.xml
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
<?xml version="1.0" encoding="UTF-8"?>
|
| 2 |
+
<project version="4">
|
| 3 |
+
<component name="StudioBotProjectSettings">
|
| 4 |
+
<option name="shareContext" value="OptedIn" />
|
| 5 |
+
</component>
|
| 6 |
+
</project>
|
.idea/vcs.xml
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
<?xml version="1.0" encoding="UTF-8"?>
|
| 2 |
+
<project version="4">
|
| 3 |
+
<component name="VcsDirectoryMappings">
|
| 4 |
+
<mapping directory="" vcs="Git" />
|
| 5 |
+
</component>
|
| 6 |
+
</project>
|
.idea/workspace.xml
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
<?xml version="1.0" encoding="UTF-8"?>
|
| 2 |
+
<project version="4">
|
| 3 |
+
<component name="AutoImportSettings">
|
| 4 |
+
<option name="autoReloadType" value="NONE" />
|
| 5 |
+
</component>
|
| 6 |
+
<component name="ChangeListManager">
|
| 7 |
+
<list default="true" id="66ee3a4c-7587-42f2-b266-bea278ef11ef" name="Changes" comment="" />
|
| 8 |
+
<option name="SHOW_DIALOG" value="false" />
|
| 9 |
+
<option name="HIGHLIGHT_CONFLICTS" value="true" />
|
| 10 |
+
<option name="HIGHLIGHT_NON_ACTIVE_CHANGELIST" value="false" />
|
| 11 |
+
<option name="LAST_RESOLUTION" value="IGNORE" />
|
| 12 |
+
</component>
|
| 13 |
+
<component name="ClangdSettings">
|
| 14 |
+
<option name="formatViaClangd" value="false" />
|
| 15 |
+
</component>
|
| 16 |
+
<component name="Git.Settings">
|
| 17 |
+
<option name="RECENT_GIT_ROOT_PATH" value="$PROJECT_DIR$" />
|
| 18 |
+
</component>
|
| 19 |
+
<component name="ProjectColorInfo"><![CDATA[{
|
| 20 |
+
"associatedIndex": 2,
|
| 21 |
+
"fromUser": false
|
| 22 |
+
}]]></component>
|
| 23 |
+
<component name="ProjectId" id="3HdoQC2fewtmSXjlJPtirHTWhuF" />
|
| 24 |
+
<component name="ProjectViewState">
|
| 25 |
+
<option name="hideEmptyMiddlePackages" value="true" />
|
| 26 |
+
<option name="showLibraryContents" value="true" />
|
| 27 |
+
</component>
|
| 28 |
+
<component name="PropertiesComponent"><![CDATA[{
|
| 29 |
+
"keyToString": {
|
| 30 |
+
"ModuleVcsDetector.initialDetectionPerformed": "true",
|
| 31 |
+
"RunOnceActivity.ShowReadmeOnStart": "true",
|
| 32 |
+
"RunOnceActivity.cidr.known.project.marker": "true",
|
| 33 |
+
"RunOnceActivity.git.unshallow": "true",
|
| 34 |
+
"RunOnceActivity.readMode.enableVisualFormatting": "true",
|
| 35 |
+
"cf.first.check.clang-format": "false",
|
| 36 |
+
"cidr.known.project.marker": "true",
|
| 37 |
+
"dart.analysis.tool.window.visible": "false",
|
| 38 |
+
"git-widget-placeholder": "main",
|
| 39 |
+
"kotlin-language-version-configured": "true",
|
| 40 |
+
"last_opened_file_path": "C:/Users/Abdo/AndroidStudioProjects/WhatsNew-Scraper",
|
| 41 |
+
"settings.editor.selected.configurable": "preferences.lookFeel",
|
| 42 |
+
"show.migrate.to.gradle.popup": "false"
|
| 43 |
+
}
|
| 44 |
+
}]]></component>
|
| 45 |
+
<component name="TaskManager">
|
| 46 |
+
<task active="true" id="Default" summary="Default task">
|
| 47 |
+
<changelist id="66ee3a4c-7587-42f2-b266-bea278ef11ef" name="Changes" comment="" />
|
| 48 |
+
<created>1786207287629</created>
|
| 49 |
+
<option name="number" value="Default" />
|
| 50 |
+
<option name="presentableId" value="Default" />
|
| 51 |
+
<updated>1786207287629</updated>
|
| 52 |
+
</task>
|
| 53 |
+
<servers />
|
| 54 |
+
</component>
|
| 55 |
+
</project>
|
__pycache__/main.cpython-314.pyc
ADDED
|
Binary file (8.17 kB). View file
|
|
|
app.py
ADDED
|
@@ -0,0 +1,269 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
WhatsNew News Intelligence Engine — Hugging Face Gradio Web Space.
|
| 3 |
+
|
| 4 |
+
Paste a news article URL (Arabic or English) or paste article text directly
|
| 5 |
+
to perform real-time NLP enrichment: Category, Sentiment, NER, Keywords,
|
| 6 |
+
Geocoding, and Metadata.
|
| 7 |
+
"""
|
| 8 |
+
|
| 9 |
+
import os
|
| 10 |
+
import re
|
| 11 |
+
import json
|
| 12 |
+
import logging
|
| 13 |
+
import requests
|
| 14 |
+
from bs4 import BeautifulSoup
|
| 15 |
+
import gradio as gr
|
| 16 |
+
|
| 17 |
+
from domain.entities.article import Article
|
| 18 |
+
from services.local_nlp_classifier import LocalNLPClassifier
|
| 19 |
+
|
| 20 |
+
logging.basicConfig(level=logging.INFO)
|
| 21 |
+
logger = logging.getLogger(__name__)
|
| 22 |
+
|
| 23 |
+
# Initialize local NLP classifier engine
|
| 24 |
+
nlp_engine = LocalNLPClassifier()
|
| 25 |
+
|
| 26 |
+
|
| 27 |
+
def scrape_article_from_url(url: str):
|
| 28 |
+
"""Scrape article title, summary, and image from a news URL."""
|
| 29 |
+
headers = {
|
| 30 |
+
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
| 31 |
+
}
|
| 32 |
+
try:
|
| 33 |
+
response = requests.get(url.strip(), headers=headers, timeout=8)
|
| 34 |
+
if response.status_code != 200:
|
| 35 |
+
return None, None, None, f"Failed to fetch URL (Status {response.status_code})"
|
| 36 |
+
|
| 37 |
+
soup = BeautifulSoup(response.text, 'html.parser')
|
| 38 |
+
|
| 39 |
+
# 1. Title Extraction
|
| 40 |
+
title = None
|
| 41 |
+
og_title = soup.find('meta', property='og:title')
|
| 42 |
+
if og_title and og_title.get('content'):
|
| 43 |
+
title = og_title['content'].strip()
|
| 44 |
+
elif soup.title and soup.title.string:
|
| 45 |
+
title = soup.title.string.strip()
|
| 46 |
+
elif soup.find('h1'):
|
| 47 |
+
title = soup.find('h1').get_text().strip()
|
| 48 |
+
|
| 49 |
+
# 2. Image Extraction
|
| 50 |
+
image_url = None
|
| 51 |
+
og_image = soup.find('meta', property='og:image') or soup.find('meta', attrs={'name': 'twitter:image'})
|
| 52 |
+
if og_image and og_image.get('content'):
|
| 53 |
+
image_url = og_image['content'].strip()
|
| 54 |
+
|
| 55 |
+
# 3. Summary / Body Extraction
|
| 56 |
+
summary = None
|
| 57 |
+
og_desc = soup.find('meta', property='og:description') or soup.find('meta', attrs={'name': 'description'})
|
| 58 |
+
if og_desc and og_desc.get('content'):
|
| 59 |
+
summary = og_desc['content'].strip()
|
| 60 |
+
else:
|
| 61 |
+
paragraphs = [p.get_text().strip() for p in soup.find_all('p') if len(p.get_text().strip()) > 30]
|
| 62 |
+
summary = ' '.join(paragraphs[:3]) if paragraphs else ""
|
| 63 |
+
|
| 64 |
+
return title, summary, image_url, None
|
| 65 |
+
except Exception as e:
|
| 66 |
+
return None, None, None, f"Error scraping URL: {str(e)}"
|
| 67 |
+
|
| 68 |
+
|
| 69 |
+
def process_nlp_analysis(url_input: str, title_input: str, summary_input: str):
|
| 70 |
+
"""Main Gradio processing handler."""
|
| 71 |
+
url = (url_input or "").strip()
|
| 72 |
+
title = (title_input or "").strip()
|
| 73 |
+
summary = (summary_input or "").strip()
|
| 74 |
+
image_url = None
|
| 75 |
+
status_msg = ""
|
| 76 |
+
|
| 77 |
+
# If URL provided, attempt web scraping first
|
| 78 |
+
if url:
|
| 79 |
+
scraped_title, scraped_summary, scraped_img, err = scrape_article_from_url(url)
|
| 80 |
+
if err:
|
| 81 |
+
status_msg = f"⚠️ {err}. Using manual text inputs."
|
| 82 |
+
else:
|
| 83 |
+
title = scraped_title or title
|
| 84 |
+
summary = scraped_summary or summary
|
| 85 |
+
image_url = scraped_img
|
| 86 |
+
status_msg = "✅ Successfully scraped article from live URL!"
|
| 87 |
+
|
| 88 |
+
if not title:
|
| 89 |
+
return (
|
| 90 |
+
"⚠️ Please enter an article URL or provide a Title & Summary.",
|
| 91 |
+
"", "", "", "", "", "", "", "", "", {}
|
| 92 |
+
)
|
| 93 |
+
|
| 94 |
+
# Detect language
|
| 95 |
+
is_arabic = any('\u0600' <= char <= '\u06FF' for char in f"{title} {summary}")
|
| 96 |
+
language = "ar" if is_arabic else "en"
|
| 97 |
+
|
| 98 |
+
# Construct Article entity
|
| 99 |
+
article = Article(
|
| 100 |
+
title=title,
|
| 101 |
+
summary=summary or title,
|
| 102 |
+
url=url or "https://whatsnew.ai/demo",
|
| 103 |
+
source="Gradio Interactive UI",
|
| 104 |
+
publishedAt=None,
|
| 105 |
+
language=language,
|
| 106 |
+
imageUrl=image_url
|
| 107 |
+
)
|
| 108 |
+
|
| 109 |
+
# Perform NLP enrichment
|
| 110 |
+
enriched = nlp_engine.enrich_article(article)
|
| 111 |
+
|
| 112 |
+
# Formatted Visual Displays
|
| 113 |
+
category_display = f"🏷️ **Category**: {enriched.category or 'General'}\n📌 **Subcategory**: {enriched.subcategory or 'N/A'}"
|
| 114 |
+
|
| 115 |
+
# Sentiment Badge
|
| 116 |
+
sent = (enriched.sentiment or "Neutral").title()
|
| 117 |
+
sent_icon = "🟢" if sent == "Positive" else "🔴" if sent == "Negative" else "⚪"
|
| 118 |
+
sentiment_display = f"{sent_icon} **Sentiment**: {sent}"
|
| 119 |
+
|
| 120 |
+
# Named Entities
|
| 121 |
+
people_str = ", ".join(enriched.people) if enriched.people else "None detected"
|
| 122 |
+
orgs_str = ", ".join(enriched.organizations) if enriched.organizations else "None detected"
|
| 123 |
+
locs_str = ", ".join(enriched.locations) if enriched.locations else "None detected"
|
| 124 |
+
ner_display = f"👤 **People**: {people_str}\n🏢 **Organizations**: {orgs_str}\n📍 **Locations**: {locs_str}"
|
| 125 |
+
|
| 126 |
+
# Geocoding
|
| 127 |
+
geo_display = f"🌍 **Region**: {enriched.region or 'Global'}\n🏳️ **Country**: {enriched.country or 'Global'}"
|
| 128 |
+
|
| 129 |
+
# Tags & Keywords
|
| 130 |
+
tags_display = " ".join([f"#{t.replace(' ', '_')}" for t in (enriched.tags or [])])
|
| 131 |
+
keywords_display = ", ".join(enriched.keywords or [])
|
| 132 |
+
|
| 133 |
+
# Stats
|
| 134 |
+
stats_display = f"⭐ **Importance Score**: {enriched.importance or 5}/10\n⏱️ **Reading Time**: {enriched.readingTime or 1} min\n🌐 **Language**: {'Arabic 🇦🇪/🇪🇬' if language == 'ar' else 'English 🇬🇧/🇺🇸'}"
|
| 135 |
+
|
| 136 |
+
# Full JSON Metadata dictionary
|
| 137 |
+
json_output = {
|
| 138 |
+
"title": enriched.title,
|
| 139 |
+
"summary": enriched.summary,
|
| 140 |
+
"url": enriched.url,
|
| 141 |
+
"imageUrl": enriched.imageUrl,
|
| 142 |
+
"language": enriched.language,
|
| 143 |
+
"category": enriched.category,
|
| 144 |
+
"subcategory": enriched.subcategory,
|
| 145 |
+
"sentiment": enriched.sentiment,
|
| 146 |
+
"region": enriched.region,
|
| 147 |
+
"country": enriched.country,
|
| 148 |
+
"people": enriched.people,
|
| 149 |
+
"organizations": enriched.organizations,
|
| 150 |
+
"locations": enriched.locations,
|
| 151 |
+
"tags": enriched.tags,
|
| 152 |
+
"keywords": enriched.keywords,
|
| 153 |
+
"importance": enriched.importance,
|
| 154 |
+
"readingTime": enriched.readingTime
|
| 155 |
+
}
|
| 156 |
+
|
| 157 |
+
return (
|
| 158 |
+
status_msg or "✅ Analysis Complete",
|
| 159 |
+
title,
|
| 160 |
+
summary,
|
| 161 |
+
category_display,
|
| 162 |
+
sentiment_display,
|
| 163 |
+
ner_display,
|
| 164 |
+
geo_display,
|
| 165 |
+
tags_display,
|
| 166 |
+
keywords_display,
|
| 167 |
+
stats_display,
|
| 168 |
+
json_output,
|
| 169 |
+
image_url or "https://images.unsplash.com/photo-1504711434969-e33886168f5c?q=80&w=1000&auto=format&fit=crop"
|
| 170 |
+
)
|
| 171 |
+
|
| 172 |
+
|
| 173 |
+
# ---------------------------------------------------------------------------
|
| 174 |
+
# Custom CSS Aesthetics
|
| 175 |
+
# ---------------------------------------------------------------------------
|
| 176 |
+
custom_css = """
|
| 177 |
+
.main-title { text-align: center; font-size: 2.2em; font-weight: bold; margin-bottom: 5px; color: #1E293B; }
|
| 178 |
+
.subtitle { text-align: center; font-size: 1.1em; color: #64748B; margin-bottom: 25px; }
|
| 179 |
+
.output-card { border-radius: 12px; padding: 15px; background: #F8FAFC; border: 1px solid #E2E8F0; }
|
| 180 |
+
"""
|
| 181 |
+
|
| 182 |
+
# ---------------------------------------------------------------------------
|
| 183 |
+
# Gradio UI Layout
|
| 184 |
+
# ---------------------------------------------------------------------------
|
| 185 |
+
with gr.Blocks(title="WhatsNew NLP Engine", css=custom_css, theme=gr.themes.Soft()) as demo:
|
| 186 |
+
gr.Markdown("<div class='main-title'>📰 WhatsNew Intelligence Engine</div>")
|
| 187 |
+
gr.Markdown("<div class='subtitle'>Real-Time Multilingual NLP Article Categorization, Sentiment Analysis, NER & Metadata Extraction</div>")
|
| 188 |
+
|
| 189 |
+
with gr.Row():
|
| 190 |
+
with gr.Column(scale=1):
|
| 191 |
+
gr.Markdown("### 📥 Article Input")
|
| 192 |
+
url_input = gr.Textbox(
|
| 193 |
+
label="Option A: Paste News Article URL (Arabic or English)",
|
| 194 |
+
placeholder="https://www.aljazeera.net/... or https://techcrunch.com/...",
|
| 195 |
+
lines=1
|
| 196 |
+
)
|
| 197 |
+
gr.Markdown("**OR Enter Manually:**")
|
| 198 |
+
title_input = gr.Textbox(
|
| 199 |
+
label="Article Title",
|
| 200 |
+
placeholder="e.g. الأهلي يفوز بلقب دوري أبطال أفريقيا",
|
| 201 |
+
lines=2
|
| 202 |
+
)
|
| 203 |
+
summary_input = gr.Textbox(
|
| 204 |
+
label="Article Summary / Text",
|
| 205 |
+
placeholder="Paste news paragraph or text summary here...",
|
| 206 |
+
lines=4
|
| 207 |
+
)
|
| 208 |
+
analyze_btn = gr.Button("🚀 Analyze & Enrich Article", variant="primary", size="lg")
|
| 209 |
+
|
| 210 |
+
gr.Markdown("---")
|
| 211 |
+
gr.Markdown("### 💡 Quick Examples")
|
| 212 |
+
gr.Examples(
|
| 213 |
+
examples=[
|
| 214 |
+
["", "الأهلي يتوج بكأس السوبر المصري بعد فوزه على الزمالك في ركلات الترجيح", "نجح الفريق الأول لكرة القدم بنادي الأهلي في تحقيق حسم بطولة كأس السوبر المصري بعد مباراة حماسية انتهت بفوزه على غريمه التقليدي الزمالك."],
|
| 215 |
+
["", "Apple unveils new M4 MacBook Pro with advanced AI capabilities", "Apple today announced its next-generation MacBook Pro powered by the M4 chip lineup, featuring enhanced neural processing for on-device artificial intelligence."],
|
| 216 |
+
["", "ارتفاع أسعار النفط العالمية بسبب التوترات الجيوسياسية في الشرق الأوسط", "شهدت أسواق النفط العالمية ارتفاعاً ملحوظاً في أسعار الخام وسط متابعة حثيثة من المستثمرين لتطورات الأوضاع الاقتصادية."]
|
| 217 |
+
],
|
| 218 |
+
inputs=[url_input, title_input, summary_input]
|
| 219 |
+
)
|
| 220 |
+
|
| 221 |
+
with gr.Column(scale=1):
|
| 222 |
+
status_box = gr.Markdown("Waiting for input...")
|
| 223 |
+
|
| 224 |
+
with gr.Tabs():
|
| 225 |
+
with gr.TabItem("📊 NLP Insights"):
|
| 226 |
+
with gr.Row():
|
| 227 |
+
category_box = gr.Markdown(elem_classes=["output-card"])
|
| 228 |
+
sentiment_box = gr.Markdown(elem_classes=["output-card"])
|
| 229 |
+
|
| 230 |
+
ner_box = gr.Markdown(elem_classes=["output-card"])
|
| 231 |
+
geo_box = gr.Markdown(elem_classes=["output-card"])
|
| 232 |
+
stats_box = gr.Markdown(elem_classes=["output-card"])
|
| 233 |
+
|
| 234 |
+
gr.Markdown("#### 🏷️ Extracted Hashtags & Tags")
|
| 235 |
+
tags_box = gr.Markdown()
|
| 236 |
+
|
| 237 |
+
gr.Markdown("#### 🔑 Key Phrases")
|
| 238 |
+
keywords_box = gr.Markdown()
|
| 239 |
+
|
| 240 |
+
with gr.TabItem("🖼️ Media & Preview"):
|
| 241 |
+
image_preview = gr.Image(label="Article Cover Image", type="url")
|
| 242 |
+
extracted_title = gr.Textbox(label="Processed Title", interactive=False)
|
| 243 |
+
extracted_summary = gr.Textbox(label="Processed Summary", interactive=False, lines=4)
|
| 244 |
+
|
| 245 |
+
with gr.TabItem("📄 Raw JSON Metadata"):
|
| 246 |
+
json_box = gr.JSON(label="Full Firestore-Ready Document Schema")
|
| 247 |
+
|
| 248 |
+
# Connect click event
|
| 249 |
+
analyze_btn.click(
|
| 250 |
+
fn=process_nlp_analysis,
|
| 251 |
+
inputs=[url_input, title_input, summary_input],
|
| 252 |
+
outputs=[
|
| 253 |
+
status_box,
|
| 254 |
+
extracted_title,
|
| 255 |
+
extracted_summary,
|
| 256 |
+
category_box,
|
| 257 |
+
sentiment_box,
|
| 258 |
+
ner_box,
|
| 259 |
+
geo_box,
|
| 260 |
+
tags_box,
|
| 261 |
+
keywords_box,
|
| 262 |
+
stats_box,
|
| 263 |
+
json_box,
|
| 264 |
+
image_preview
|
| 265 |
+
]
|
| 266 |
+
)
|
| 267 |
+
|
| 268 |
+
if __name__ == "__main__":
|
| 269 |
+
demo.launch()
|
data/__pycache__/firestore_repo.cpython-314.pyc
ADDED
|
Binary file (3.13 kB). View file
|
|
|
data/repositories/__pycache__/article_repository_impl.cpython-314.pyc
ADDED
|
Binary file (3.95 kB). View file
|
|
|
data/repositories/article_repository_impl.py
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import firebase_admin
|
| 2 |
+
from firebase_admin import credentials, firestore
|
| 3 |
+
from typing import List
|
| 4 |
+
import logging
|
| 5 |
+
from domain.entities.article import Article
|
| 6 |
+
from domain.repositories.article_repository import ArticleRepository
|
| 7 |
+
|
| 8 |
+
logger = logging.getLogger(__name__)
|
| 9 |
+
|
| 10 |
+
class FirestoreArticleRepository(ArticleRepository):
|
| 11 |
+
def __init__(self, service_account_path: str = "service-account.json"):
|
| 12 |
+
try:
|
| 13 |
+
if not firebase_admin._apps:
|
| 14 |
+
cred = credentials.Certificate(service_account_path)
|
| 15 |
+
firebase_admin.initialize_app(cred)
|
| 16 |
+
self.db = firestore.client()
|
| 17 |
+
except Exception as e:
|
| 18 |
+
logger.error(f"Failed to initialize Firebase Admin: {e}")
|
| 19 |
+
self.db = None
|
| 20 |
+
|
| 21 |
+
def save_articles(self, articles: List[Article]) -> List[Article]:
|
| 22 |
+
if not self.db:
|
| 23 |
+
logger.error("Firestore DB not initialized. Cannot save articles.")
|
| 24 |
+
return []
|
| 25 |
+
|
| 26 |
+
articles_ref = self.db.collection('articles')
|
| 27 |
+
new_articles = []
|
| 28 |
+
for article in articles:
|
| 29 |
+
try:
|
| 30 |
+
# Use the generated deterministic ID as the document ID
|
| 31 |
+
doc_id = article.id
|
| 32 |
+
doc_ref = articles_ref.document(doc_id)
|
| 33 |
+
|
| 34 |
+
# Check if document exists
|
| 35 |
+
doc_snap = doc_ref.get()
|
| 36 |
+
if not doc_snap.exists:
|
| 37 |
+
doc_ref.set(article.model_dump(mode='json'))
|
| 38 |
+
new_articles.append(article)
|
| 39 |
+
else:
|
| 40 |
+
# Backfill/Update image if existing document was saved without an image
|
| 41 |
+
existing_data = doc_snap.to_dict() or {}
|
| 42 |
+
if not existing_data.get('imageUrl') and article.imageUrl:
|
| 43 |
+
logger.info(f"Backfilling missing image for existing article: {article.title[:40]}")
|
| 44 |
+
doc_ref.update({
|
| 45 |
+
"imageUrl": article.imageUrl,
|
| 46 |
+
"hasImage": True
|
| 47 |
+
})
|
| 48 |
+
except Exception as e:
|
| 49 |
+
logger.error(f"Failed to save article {article.title}: {e}")
|
| 50 |
+
|
| 51 |
+
logger.info(f"Saved {len(new_articles)} new articles to Firestore.")
|
| 52 |
+
return new_articles
|
| 53 |
+
|
| 54 |
+
def exists(self, article_id: str) -> bool:
|
| 55 |
+
if not self.db:
|
| 56 |
+
return False
|
| 57 |
+
doc_ref = self.db.collection('articles').document(article_id)
|
| 58 |
+
return doc_ref.get().exists
|
domain/__pycache__/interfaces.cpython-314.pyc
ADDED
|
Binary file (1.21 kB). View file
|
|
|
domain/__pycache__/models.cpython-314.pyc
ADDED
|
Binary file (2.04 kB). View file
|
|
|
domain/entities/__pycache__/article.cpython-314.pyc
ADDED
|
Binary file (5.69 kB). View file
|
|
|
domain/entities/article.py
ADDED
|
@@ -0,0 +1,74 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from pydantic import BaseModel, HttpUrl, Field, field_validator
|
| 2 |
+
from typing import Optional, List
|
| 3 |
+
import datetime
|
| 4 |
+
import uuid
|
| 5 |
+
import hashlib
|
| 6 |
+
|
| 7 |
+
class Article(BaseModel):
|
| 8 |
+
id: str = Field(default_factory=lambda: str(uuid.uuid4()))
|
| 9 |
+
title: str
|
| 10 |
+
summary: str
|
| 11 |
+
url: str
|
| 12 |
+
imageUrl: Optional[str] = None
|
| 13 |
+
source: str
|
| 14 |
+
publishedAt: datetime.datetime
|
| 15 |
+
language: str = "ar"
|
| 16 |
+
|
| 17 |
+
# Enriched Fields
|
| 18 |
+
category: Optional[str] = None
|
| 19 |
+
subcategory: Optional[str] = None
|
| 20 |
+
region: Optional[str] = None
|
| 21 |
+
country: Optional[str] = None
|
| 22 |
+
city: Optional[str] = None
|
| 23 |
+
|
| 24 |
+
tags: List[str] = Field(default_factory=list)
|
| 25 |
+
keywords: List[str] = Field(default_factory=list)
|
| 26 |
+
people: List[str] = Field(default_factory=list)
|
| 27 |
+
organizations: List[str] = Field(default_factory=list)
|
| 28 |
+
locations: List[str] = Field(default_factory=list)
|
| 29 |
+
|
| 30 |
+
sentiment: Optional[str] = None
|
| 31 |
+
importance: Optional[int] = None
|
| 32 |
+
readingTime: Optional[int] = None
|
| 33 |
+
hasImage: bool = False
|
| 34 |
+
|
| 35 |
+
@classmethod
|
| 36 |
+
def normalize_url(cls, url: str) -> str:
|
| 37 |
+
"""Strips http/https differences, www prefix, and tracking parameters to ensure 100% identical URL hashing."""
|
| 38 |
+
from urllib.parse import urlparse, parse_qs, urlunparse, urlencode
|
| 39 |
+
# Force https and strip www.
|
| 40 |
+
url_clean_scheme = url.replace('http://', 'https://')
|
| 41 |
+
parsed = urlparse(url_clean_scheme)
|
| 42 |
+
netloc = parsed.netloc.lower()
|
| 43 |
+
if netloc.startswith('www.'):
|
| 44 |
+
netloc = netloc[4:]
|
| 45 |
+
|
| 46 |
+
# Strip tracking query params
|
| 47 |
+
query_params = parse_qs(parsed.query, keep_blank_values=True)
|
| 48 |
+
filtered_params = {k: v for k, v in query_params.items() if not k.startswith('utm_') and k not in ['ref', 'fbclid', 'rss', 'src', 'amp']}
|
| 49 |
+
clean_query = urlencode(filtered_params, doseq=True)
|
| 50 |
+
# Reconstruct URL without trailing slash and tracking params
|
| 51 |
+
clean_url = urlunparse((parsed.scheme, netloc, parsed.path.rstrip('/'), parsed.params, clean_query, ''))
|
| 52 |
+
return clean_url
|
| 53 |
+
|
| 54 |
+
@classmethod
|
| 55 |
+
def generate_id_from_url(cls, url: str) -> str:
|
| 56 |
+
"""Generates a deterministic ID from the normalized URL to prevent duplicates."""
|
| 57 |
+
clean_url = cls.normalize_url(url)
|
| 58 |
+
return hashlib.md5(clean_url.encode('utf-8')).hexdigest()
|
| 59 |
+
|
| 60 |
+
@classmethod
|
| 61 |
+
def generate_title_hash(cls, title: str) -> str:
|
| 62 |
+
"""Generates a normalized hash of the title to detect cross-source headline duplicates."""
|
| 63 |
+
import re
|
| 64 |
+
clean_title = re.sub(r'[^\w\s]', '', title.lower()).strip()
|
| 65 |
+
return hashlib.md5(clean_title.encode('utf-8')).hexdigest()
|
| 66 |
+
|
| 67 |
+
@field_validator('readingTime', 'importance', mode='before')
|
| 68 |
+
def convert_float_to_int(cls, v):
|
| 69 |
+
if v is not None:
|
| 70 |
+
try:
|
| 71 |
+
return max(1, round(float(v)))
|
| 72 |
+
except (ValueError, TypeError):
|
| 73 |
+
return v
|
| 74 |
+
return v
|
domain/interfaces.py
ADDED
|
@@ -0,0 +1,10 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from typing import List, Protocol
|
| 2 |
+
from domain.entities.article import Article
|
| 3 |
+
|
| 4 |
+
class ScraperInterface(Protocol):
|
| 5 |
+
source_name: str
|
| 6 |
+
language: str
|
| 7 |
+
|
| 8 |
+
def scrape(self) -> List[Article]:
|
| 9 |
+
"""Scrapes the target website and returns a list of Articles."""
|
| 10 |
+
...
|
domain/repositories/__pycache__/article_repository.cpython-314.pyc
ADDED
|
Binary file (1.53 kB). View file
|
|
|
domain/repositories/article_repository.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from abc import ABC, abstractmethod
|
| 2 |
+
from typing import List
|
| 3 |
+
from domain.entities.article import Article
|
| 4 |
+
|
| 5 |
+
class ArticleRepository(ABC):
|
| 6 |
+
@abstractmethod
|
| 7 |
+
def save_articles(self, articles: List[Article]) -> List[Article]:
|
| 8 |
+
"""Saves a list of articles and returns the list of newly saved articles."""
|
| 9 |
+
pass
|
| 10 |
+
|
| 11 |
+
@abstractmethod
|
| 12 |
+
def exists(self, article_id: str) -> bool:
|
| 13 |
+
"""Checks if an article already exists by its ID."""
|
| 14 |
+
pass
|
domain/usecases/__pycache__/scrape_and_enrich.cpython-314.pyc
ADDED
|
Binary file (4.29 kB). View file
|
|
|
domain/usecases/scrape_and_enrich.py
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import logging
|
| 2 |
+
from typing import List
|
| 3 |
+
from domain.entities.article import Article
|
| 4 |
+
from domain.repositories.article_repository import ArticleRepository
|
| 5 |
+
from domain.interfaces import ScraperInterface
|
| 6 |
+
from services.ai_classifier import HuggingFaceClassifier
|
| 7 |
+
from services.duplicate_detector import DuplicateDetector
|
| 8 |
+
|
| 9 |
+
from services.lexicons import CATEGORY_IMAGES_MAP
|
| 10 |
+
|
| 11 |
+
logger = logging.getLogger(__name__)
|
| 12 |
+
|
| 13 |
+
class ScrapeAndEnrichUseCase:
|
| 14 |
+
def __init__(self,
|
| 15 |
+
repository: ArticleRepository,
|
| 16 |
+
scrapers: List[ScraperInterface],
|
| 17 |
+
classifier: HuggingFaceClassifier,
|
| 18 |
+
duplicate_detector: DuplicateDetector):
|
| 19 |
+
self.repository = repository
|
| 20 |
+
self.scrapers = scrapers
|
| 21 |
+
self.classifier = classifier
|
| 22 |
+
self.duplicate_detector = duplicate_detector
|
| 23 |
+
|
| 24 |
+
def execute(self) -> List[Article]:
|
| 25 |
+
all_articles = []
|
| 26 |
+
|
| 27 |
+
# 1. Scrape raw articles from all sources
|
| 28 |
+
for scraper in self.scrapers:
|
| 29 |
+
logger.info(f"Scraping raw feed from {scraper.source_name} ({scraper.language})...")
|
| 30 |
+
try:
|
| 31 |
+
articles = scraper.scrape()
|
| 32 |
+
logger.info(f"Found {len(articles)} raw articles from {scraper.source_name}")
|
| 33 |
+
all_articles.extend(articles)
|
| 34 |
+
except Exception as e:
|
| 35 |
+
logger.error(f"Failed to scrape {scraper.source_name}: {e}")
|
| 36 |
+
|
| 37 |
+
if not all_articles:
|
| 38 |
+
logger.info("No articles found across all scrapers.")
|
| 39 |
+
return []
|
| 40 |
+
|
| 41 |
+
# 2. Filter duplicates
|
| 42 |
+
new_articles = self.duplicate_detector.filter_new_articles(all_articles)
|
| 43 |
+
if not new_articles:
|
| 44 |
+
logger.info("No new articles to enrich.")
|
| 45 |
+
return []
|
| 46 |
+
|
| 47 |
+
# 3. Enrich articles via AI pipeline & apply Layer 3 Image Fallback
|
| 48 |
+
enriched_articles = []
|
| 49 |
+
logger.info(f"Starting AI enrichment for {len(new_articles)} new articles...")
|
| 50 |
+
|
| 51 |
+
for index, article in enumerate(new_articles):
|
| 52 |
+
logger.info(f"Enriching article {index + 1}/{len(new_articles)}: {article.title[:50]}...")
|
| 53 |
+
enriched = self.classifier.enrich_article(article)
|
| 54 |
+
|
| 55 |
+
# Layer 3: Ensure imageUrl is NEVER null
|
| 56 |
+
if not enriched.imageUrl:
|
| 57 |
+
fallback_img = CATEGORY_IMAGES_MAP.get(enriched.category, CATEGORY_IMAGES_MAP["General"])
|
| 58 |
+
enriched.imageUrl = fallback_img
|
| 59 |
+
enriched.hasImage = True
|
| 60 |
+
else:
|
| 61 |
+
enriched.hasImage = True
|
| 62 |
+
|
| 63 |
+
enriched_articles.append(enriched)
|
| 64 |
+
|
| 65 |
+
# 4. Save to Repository
|
| 66 |
+
logger.info(f"Saving {len(enriched_articles)} enriched articles to database...")
|
| 67 |
+
new_saved_articles = self.repository.save_articles(enriched_articles)
|
| 68 |
+
|
| 69 |
+
return new_saved_articles
|
main.py
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import logging
|
| 2 |
+
from scrapers.rss_scraper import RssScraper
|
| 3 |
+
from data.repositories.article_repository_impl import FirestoreArticleRepository
|
| 4 |
+
from services.ai_classifier import HuggingFaceClassifier
|
| 5 |
+
from services.duplicate_detector import DuplicateDetector
|
| 6 |
+
from domain.usecases.scrape_and_enrich import ScrapeAndEnrichUseCase
|
| 7 |
+
|
| 8 |
+
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(name)s - %(levelname)s - %(message)s')
|
| 9 |
+
logger = logging.getLogger(__name__)
|
| 10 |
+
|
| 11 |
+
# Configure our RSS sources across diverse categories (General, Tech, Gaming, Food, Sports, Business, Science, Entertainment)
|
| 12 |
+
SOURCES = [
|
| 13 |
+
# --- World & Middle East News (English) ---
|
| 14 |
+
RssScraper("BBC Middle East", "en", "http://feeds.bbci.co.uk/news/world/middle_east/rss.xml", category="Middle East"),
|
| 15 |
+
RssScraper("BBC World", "en", "http://feeds.bbci.co.uk/news/world/rss.xml", category="World"),
|
| 16 |
+
RssScraper("Middle East Eye", "en", "https://www.middleeasteye.net/rss", category="Middle East"),
|
| 17 |
+
RssScraper("Middle East Monitor", "en", "https://www.middleeastmonitor.com/feed/", category="Middle East"),
|
| 18 |
+
RssScraper("The Guardian Middle East", "en", "https://www.theguardian.com/world/middleeast/rss", category="Middle East"),
|
| 19 |
+
RssScraper("NY Times Middle East", "en", "https://rss.nytimes.com/services/xml/rss/nyt/MiddleEast.xml", category="Middle East"),
|
| 20 |
+
RssScraper("Al Jazeera English", "en", "https://www.aljazeera.com/xml/rss/all.xml", category="World"),
|
| 21 |
+
|
| 22 |
+
# --- Egyptian & Arab World News (Arabic) ---
|
| 23 |
+
RssScraper("Youm7", "ar", "http://www.youm7.com/rss/SectionRss?SectionID=65", category="Egypt"),
|
| 24 |
+
RssScraper("Al-Ahram", "ar", "https://gate.ahram.org.eg/NewsRss.aspx", category="Egypt"),
|
| 25 |
+
RssScraper("Al Masry Al Youm", "ar", "https://www.almasryalyoum.com/rss/rss", category="Egypt"),
|
| 26 |
+
RssScraper("Sada El Balad", "ar", "https://www.elbalad.news/rss.aspx", category="Egypt"),
|
| 27 |
+
RssScraper("Al Shorouk", "ar", "https://www.shorouknews.com/rss/news.aspx", category="Egypt"),
|
| 28 |
+
RssScraper("Al-Dustour", "ar", "https://www.dostor.org/rss.aspx", category="Egypt"),
|
| 29 |
+
RssScraper("Sky News Arabia", "ar", "https://www.skynewsarabia.com/rss", category="Middle East"),
|
| 30 |
+
RssScraper("RT Arabic", "ar", "https://arabic.rt.com/rss/", category="Middle East"),
|
| 31 |
+
RssScraper("Asharq Al-Awsat", "ar", "https://aawsat.com/feed", category="Middle East"),
|
| 32 |
+
RssScraper("Alhurra", "ar", "https://www.alhurra.com/rss", category="Middle East"),
|
| 33 |
+
RssScraper("BBC Arabic", "ar", "http://feeds.bbci.co.uk/arabic/rss.xml", category="Middle East"),
|
| 34 |
+
RssScraper("CNN Arabic", "ar", "https://arabic.cnn.com/api/v1/rss/middle_east/rss.xml", category="Middle East"),
|
| 35 |
+
RssScraper("France 24 Arabic", "ar", "https://www.france24.com/ar/rss", category="Middle East"),
|
| 36 |
+
RssScraper("Hespress", "ar", "https://www.hespress.com/feed", category="North Africa"),
|
| 37 |
+
RssScraper("Echorouk", "ar", "https://www.echoroukonline.com/feed", category="North Africa"),
|
| 38 |
+
|
| 39 |
+
# --- Technology & AI ---
|
| 40 |
+
RssScraper("TechCrunch", "en", "https://techcrunch.com/feed/", category="Technology"),
|
| 41 |
+
RssScraper("Wired", "en", "https://www.wired.com/feed/rss", category="Technology"),
|
| 42 |
+
RssScraper("The Verge", "en", "https://www.theverge.com/rss/index.xml", category="Technology"),
|
| 43 |
+
RssScraper("Ars Technica", "en", "https://feeds.arstechnica.com/arstechnica/index", category="Technology"),
|
| 44 |
+
RssScraper("Engadget", "en", "https://www.engadget.com/rss.xml", category="Technology"),
|
| 45 |
+
RssScraper("Aitnews", "ar", "https://aitnews.com/feed/", category="Technology"),
|
| 46 |
+
RssScraper("Unlimit Tech", "ar", "https://unlimit-tech.com/feed/", category="Technology"),
|
| 47 |
+
|
| 48 |
+
# --- Gaming & Esports ---
|
| 49 |
+
RssScraper("IGN", "en", "https://feeds.feedburner.com/ign/all", category="Gaming"),
|
| 50 |
+
RssScraper("Eurogamer", "en", "https://www.eurogamer.net/?format=rss", category="Gaming"),
|
| 51 |
+
RssScraper("GameSpot", "en", "https://www.gamespot.com/feeds/news/", category="Gaming"),
|
| 52 |
+
RssScraper("Polygon", "en", "https://www.polygon.com/rss/index.xml", category="Gaming"),
|
| 53 |
+
RssScraper("Saudi Gamer", "ar", "https://saudigamer.com/feed/", category="Gaming"),
|
| 54 |
+
RssScraper("TrueGaming", "ar", "https://www.true-gaming.net/home/feed/", category="Gaming"),
|
| 55 |
+
|
| 56 |
+
# --- Food, Cooking & Lifestyle ---
|
| 57 |
+
RssScraper("Serious Eats", "en", "https://www.seriouseats.com/feed", category="Food"),
|
| 58 |
+
RssScraper("BBC Good Food", "en", "https://www.bbcgoodfood.com/feed/rss2", category="Food"),
|
| 59 |
+
RssScraper("Food52", "en", "https://food52.com/blog.rss", category="Food"),
|
| 60 |
+
RssScraper("SuperMama", "ar", "https://www.supermama.me/rss.xml", category="Lifestyle"),
|
| 61 |
+
|
| 62 |
+
# --- Business, Economy & Finance ---
|
| 63 |
+
RssScraper("CNBC World", "en", "https://www.cnbc.com/id/100727362/device/rss/rss.html", category="Business"),
|
| 64 |
+
RssScraper("Financial Times", "en", "https://www.ft.com/?format=rss", category="Business"),
|
| 65 |
+
RssScraper("RT Arabic Business", "ar", "https://arabic.rt.com/rss/business/", category="Business"),
|
| 66 |
+
RssScraper("Al Mal News", "ar", "https://almalnews.com/feed/", category="Business"),
|
| 67 |
+
|
| 68 |
+
# --- Sports ---
|
| 69 |
+
RssScraper("BBC Sport", "en", "http://feeds.bbci.co.uk/sport/rss.xml", category="Sports"),
|
| 70 |
+
RssScraper("RT Arabic Sports", "ar", "https://arabic.rt.com/rss/sport/", category="Sports"),
|
| 71 |
+
RssScraper("FilGoal", "ar", "https://www.filgoal.com/rss/news", category="Sports"),
|
| 72 |
+
|
| 73 |
+
# --- Science & Space ---
|
| 74 |
+
RssScraper("ScienceDaily", "en", "https://www.sciencedaily.com/rss/all.xml", category="Science"),
|
| 75 |
+
RssScraper("NASA News", "en", "https://www.nasa.gov/news-release/feed/", category="Science"),
|
| 76 |
+
RssScraper("RT Arabic Science", "ar", "https://arabic.rt.com/rss/space/", category="Science"),
|
| 77 |
+
|
| 78 |
+
# --- Entertainment & Culture ---
|
| 79 |
+
RssScraper("Variety", "en", "https://variety.com/feed/", category="Entertainment"),
|
| 80 |
+
RssScraper("Hollywood Reporter", "en", "https://www.hollywoodreporter.com/feed/", category="Entertainment")
|
| 81 |
+
]
|
| 82 |
+
|
| 83 |
+
def job():
|
| 84 |
+
logger.info("Starting scraping job with AI Enrichment...")
|
| 85 |
+
|
| 86 |
+
# 1. Initialize dependencies
|
| 87 |
+
try:
|
| 88 |
+
repo = FirestoreArticleRepository(service_account_path="service-account.json")
|
| 89 |
+
except Exception as e:
|
| 90 |
+
logger.error(f"Failed to initialize Firestore: {e}")
|
| 91 |
+
return
|
| 92 |
+
|
| 93 |
+
if not repo.db:
|
| 94 |
+
logger.error("Skipping job because Firestore is not initialized. Please ensure service-account.json exists.")
|
| 95 |
+
return
|
| 96 |
+
|
| 97 |
+
classifier = HuggingFaceClassifier()
|
| 98 |
+
duplicate_detector = DuplicateDetector(repository=repo)
|
| 99 |
+
|
| 100 |
+
# 2. Setup Usecase
|
| 101 |
+
usecase = ScrapeAndEnrichUseCase(
|
| 102 |
+
repository=repo,
|
| 103 |
+
scrapers=SOURCES,
|
| 104 |
+
classifier=classifier,
|
| 105 |
+
duplicate_detector=duplicate_detector
|
| 106 |
+
)
|
| 107 |
+
|
| 108 |
+
# 3. Execute Pipeline
|
| 109 |
+
new_articles = usecase.execute()
|
| 110 |
+
|
| 111 |
+
# 4. Send Notifications
|
| 112 |
+
if new_articles:
|
| 113 |
+
try:
|
| 114 |
+
from firebase_admin import messaging
|
| 115 |
+
top_article = new_articles[0]
|
| 116 |
+
message = messaging.Message(
|
| 117 |
+
notification=messaging.Notification(
|
| 118 |
+
title=f"New story from {top_article.source}",
|
| 119 |
+
body=top_article.title
|
| 120 |
+
),
|
| 121 |
+
topic='daily_news'
|
| 122 |
+
)
|
| 123 |
+
response = messaging.send(message)
|
| 124 |
+
logger.info(f"Successfully sent FCM notification: {response}")
|
| 125 |
+
except Exception as e:
|
| 126 |
+
logger.error(f"Failed to send FCM notification: {e}")
|
| 127 |
+
else:
|
| 128 |
+
logger.info("No new articles processed.")
|
| 129 |
+
|
| 130 |
+
if __name__ == "__main__":
|
| 131 |
+
logger.info("News Intelligence Engine started.")
|
| 132 |
+
job()
|
| 133 |
+
logger.info("Scraping job finished.")
|
nlp_training/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
# NLP Training Scripts for WhatsNew News Intelligence Engine
|
nlp_training/prepare_datasets.py
ADDED
|
@@ -0,0 +1,398 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Dataset Preparation Utility for WhatsNew NLP Training.
|
| 3 |
+
|
| 4 |
+
Downloads and prepares Arabic + English datasets for fine-tuning:
|
| 5 |
+
- SANAD + HuffPost → Category Classification
|
| 6 |
+
- ASAD + SST-2 → Sentiment Analysis
|
| 7 |
+
- WikiANN → Named Entity Recognition
|
| 8 |
+
|
| 9 |
+
Usage:
|
| 10 |
+
python nlp_training/prepare_datasets.py --task all
|
| 11 |
+
python nlp_training/prepare_datasets.py --task category
|
| 12 |
+
python nlp_training/prepare_datasets.py --task sentiment
|
| 13 |
+
python nlp_training/prepare_datasets.py --task ner
|
| 14 |
+
"""
|
| 15 |
+
|
| 16 |
+
import os
|
| 17 |
+
import argparse
|
| 18 |
+
import json
|
| 19 |
+
import logging
|
| 20 |
+
|
| 21 |
+
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
|
| 22 |
+
logger = logging.getLogger(__name__)
|
| 23 |
+
|
| 24 |
+
# Our unified 13 categories (must match CATEGORIES_MAP keys)
|
| 25 |
+
UNIFIED_CATEGORIES = [
|
| 26 |
+
"Sports", "Technology", "Gaming", "Business", "Health", "Science",
|
| 27 |
+
"Food", "Entertainment", "Politics", "Education", "Crime", "Weather", "Travel"
|
| 28 |
+
]
|
| 29 |
+
|
| 30 |
+
# -----------------------------------------------------------------------
|
| 31 |
+
# Label mapping: Map source dataset labels → our unified category labels
|
| 32 |
+
# -----------------------------------------------------------------------
|
| 33 |
+
|
| 34 |
+
# SANAD Arabic dataset label → our category
|
| 35 |
+
SANAD_LABEL_MAP = {
|
| 36 |
+
"culture": "Entertainment",
|
| 37 |
+
"finance": "Business",
|
| 38 |
+
"medical": "Health",
|
| 39 |
+
"politics": "Politics",
|
| 40 |
+
"religion": "Education",
|
| 41 |
+
"sports": "Sports",
|
| 42 |
+
"tech": "Technology",
|
| 43 |
+
}
|
| 44 |
+
|
| 45 |
+
# HuffPost English dataset label → our category
|
| 46 |
+
HUFFPOST_LABEL_MAP = {
|
| 47 |
+
"POLITICS": "Politics",
|
| 48 |
+
"WELLNESS": "Health",
|
| 49 |
+
"ENTERTAINMENT": "Entertainment",
|
| 50 |
+
"TRAVEL": "Travel",
|
| 51 |
+
"STYLE & BEAUTY": "Entertainment",
|
| 52 |
+
"PARENTING": "Education",
|
| 53 |
+
"HEALTHY LIVING": "Health",
|
| 54 |
+
"QUEER VOICES": "Entertainment",
|
| 55 |
+
"FOOD & DRINK": "Food",
|
| 56 |
+
"BUSINESS": "Business",
|
| 57 |
+
"COMEDY": "Entertainment",
|
| 58 |
+
"SPORTS": "Sports",
|
| 59 |
+
"BLACK VOICES": "Politics",
|
| 60 |
+
"HOME & LIVING": "Food",
|
| 61 |
+
"PARENTS": "Education",
|
| 62 |
+
"THE WORLDPOST": "Politics",
|
| 63 |
+
"WEDDINGS": "Entertainment",
|
| 64 |
+
"WOMEN": "Politics",
|
| 65 |
+
"IMPACT": "Politics",
|
| 66 |
+
"DIVORCE": "Entertainment",
|
| 67 |
+
"CRIME": "Crime",
|
| 68 |
+
"MEDIA": "Entertainment",
|
| 69 |
+
"WEIRD NEWS": "Entertainment",
|
| 70 |
+
"GREEN": "Science",
|
| 71 |
+
"WORLDPOST": "Politics",
|
| 72 |
+
"RELIGION": "Education",
|
| 73 |
+
"SCIENCE": "Science",
|
| 74 |
+
"TECH": "Technology",
|
| 75 |
+
"TASTE": "Food",
|
| 76 |
+
"MONEY": "Business",
|
| 77 |
+
"ARTS": "Entertainment",
|
| 78 |
+
"FIFTY": "Entertainment",
|
| 79 |
+
"GOOD NEWS": "Entertainment",
|
| 80 |
+
"ARTS & CULTURE": "Entertainment",
|
| 81 |
+
"ENVIRONMENT": "Science",
|
| 82 |
+
"COLLEGE": "Education",
|
| 83 |
+
"LATINO VOICES": "Politics",
|
| 84 |
+
"CULTURE & ARTS": "Entertainment",
|
| 85 |
+
"EDUCATION": "Education",
|
| 86 |
+
"STYLE": "Entertainment",
|
| 87 |
+
}
|
| 88 |
+
|
| 89 |
+
SENTIMENT_LABELS = ["Positive", "Negative", "Neutral"]
|
| 90 |
+
|
| 91 |
+
|
| 92 |
+
def prepare_category_datasets(output_dir: str):
|
| 93 |
+
"""
|
| 94 |
+
Download and prepare news category classification datasets (Arabic + English).
|
| 95 |
+
Outputs: category_train.jsonl, category_test.jsonl with {text, label, language} pairs.
|
| 96 |
+
"""
|
| 97 |
+
from datasets import load_dataset
|
| 98 |
+
|
| 99 |
+
os.makedirs(output_dir, exist_ok=True)
|
| 100 |
+
raw_dir = "nlp_training/raw_data"
|
| 101 |
+
os.makedirs(raw_dir, exist_ok=True)
|
| 102 |
+
all_samples = []
|
| 103 |
+
|
| 104 |
+
# --- Local Kaggle Raw Data Check ---
|
| 105 |
+
for filename in os.listdir(raw_dir):
|
| 106 |
+
filepath = os.path.join(raw_dir, filename)
|
| 107 |
+
if filename.endswith(".csv"):
|
| 108 |
+
try:
|
| 109 |
+
import pandas as pd
|
| 110 |
+
df = pd.read_csv(filepath)
|
| 111 |
+
text_col = next((c for c in df.columns if c.lower() in ["text", "article", "content", "headline"]), None)
|
| 112 |
+
label_col = next((c for c in df.columns if c.lower() in ["label", "category", "class", "topic"]), None)
|
| 113 |
+
if text_col and label_col:
|
| 114 |
+
loaded = 0
|
| 115 |
+
for _, row in df.iterrows():
|
| 116 |
+
text = str(row[text_col])
|
| 117 |
+
raw_label = str(row[label_col]).lower()
|
| 118 |
+
mapped = SANAD_LABEL_MAP.get(raw_label) or HUFFPOST_LABEL_MAP.get(raw_label.upper())
|
| 119 |
+
if mapped and text:
|
| 120 |
+
lang = "ar" if any('\u0600' <= char <= '\u06FF' for char in text[:100]) else "en"
|
| 121 |
+
all_samples.append({"text": text[:512], "label": mapped, "language": lang})
|
| 122 |
+
loaded += 1
|
| 123 |
+
logger.info(f"Loaded {loaded} samples from local CSV '{filename}'")
|
| 124 |
+
except Exception as e:
|
| 125 |
+
logger.warning(f"Failed parsing local file {filename}: {e}")
|
| 126 |
+
|
| 127 |
+
# --- English: AG News ---
|
| 128 |
+
logger.info("Loading AG News English dataset...")
|
| 129 |
+
ag_news_map = {0: "Politics", 1: "Sports", 2: "Business", 3: "Technology"}
|
| 130 |
+
try:
|
| 131 |
+
ag_ds = load_dataset("ag_news", split="train")
|
| 132 |
+
en_count = 0
|
| 133 |
+
for sample in ag_ds:
|
| 134 |
+
text = sample.get("text", "")
|
| 135 |
+
label_id = sample.get("label")
|
| 136 |
+
mapped = ag_news_map.get(label_id)
|
| 137 |
+
if mapped and text:
|
| 138 |
+
all_samples.append({"text": text[:512], "label": mapped, "language": "en"})
|
| 139 |
+
en_count += 1
|
| 140 |
+
logger.info(f"Loaded {en_count} English samples from AG News")
|
| 141 |
+
except Exception as e:
|
| 142 |
+
logger.warning(f"Could not load AG News: {e}")
|
| 143 |
+
|
| 144 |
+
# --- English: 20 Newsgroups ---
|
| 145 |
+
try:
|
| 146 |
+
hp_ds = load_dataset("SetFit/20_newsgroups", split="train")
|
| 147 |
+
newsgroups_map = {
|
| 148 |
+
"rec.sport.baseball": "Sports", "rec.sport.hockey": "Sports",
|
| 149 |
+
"comp.graphics": "Technology", "comp.sys.mac.hardware": "Technology",
|
| 150 |
+
"talk.politics.mideast": "Politics", "talk.politics.misc": "Politics",
|
| 151 |
+
"sci.space": "Science", "sci.med": "Health"
|
| 152 |
+
}
|
| 153 |
+
for sample in hp_ds:
|
| 154 |
+
text = sample.get("text", "")
|
| 155 |
+
label_text = sample.get("label_text", "")
|
| 156 |
+
mapped = newsgroups_map.get(label_text)
|
| 157 |
+
if mapped and text:
|
| 158 |
+
all_samples.append({"text": text[:512], "label": mapped, "language": "en"})
|
| 159 |
+
except Exception:
|
| 160 |
+
pass
|
| 161 |
+
|
| 162 |
+
# --- Arabic: Arabic News Datasets ---
|
| 163 |
+
logger.info("Loading Arabic news classification dataset...")
|
| 164 |
+
arabic_loaded = False
|
| 165 |
+
arabic_ds_candidates = ["arbml/Arabic_News_Tweets", "he3als/sanad", "ahedag/arabic_news"]
|
| 166 |
+
|
| 167 |
+
for candidate in arabic_ds_candidates:
|
| 168 |
+
try:
|
| 169 |
+
ar_ds = load_dataset(candidate, split="train")
|
| 170 |
+
ar_count = 0
|
| 171 |
+
for sample in ar_ds:
|
| 172 |
+
text = sample.get("text") or sample.get("article") or sample.get("tweet") or ""
|
| 173 |
+
raw_label = str(sample.get("label") or sample.get("category") or "").lower()
|
| 174 |
+
mapped = SANAD_LABEL_MAP.get(raw_label)
|
| 175 |
+
if not mapped:
|
| 176 |
+
if any(k in raw_label for k in ["رياض", "sport"]): mapped = "Sports"
|
| 177 |
+
elif any(k in raw_label for k in ["تقن", "tech"]): mapped = "Technology"
|
| 178 |
+
elif any(k in raw_label for k in ["سياس", "politic"]): mapped = "Politics"
|
| 179 |
+
elif any(k in raw_label for k in ["اقتصاد", "مال", "econ"]): mapped = "Business"
|
| 180 |
+
elif any(k in raw_label for k in ["صح", "health"]): mapped = "Health"
|
| 181 |
+
elif any(k in raw_label for k in ["فن", "ثقاف", "entertain"]): mapped = "Entertainment"
|
| 182 |
+
|
| 183 |
+
if mapped and text:
|
| 184 |
+
all_samples.append({"text": text[:512], "label": mapped, "language": "ar"})
|
| 185 |
+
ar_count += 1
|
| 186 |
+
if ar_count > 0:
|
| 187 |
+
logger.info(f"Loaded {ar_count} Arabic news samples from '{candidate}'")
|
| 188 |
+
arabic_loaded = True
|
| 189 |
+
break
|
| 190 |
+
except Exception:
|
| 191 |
+
continue
|
| 192 |
+
|
| 193 |
+
if not arabic_loaded and len([s for s in all_samples if s["language"] == "ar"]) < 10:
|
| 194 |
+
logger.info("Using built-in Arabic news samples generator for fallback testing...")
|
| 195 |
+
sample_ar_news = [
|
| 196 |
+
("الأهلي يفوز على الزمالك في قمة الدوري الممتاز", "Sports"),
|
| 197 |
+
("إطلاق جديد لهاتف آيفون يدعم تقنيات الذكاء الاصطناعي", "Technology"),
|
| 198 |
+
("ارتفاع أسعار النفط والأسهم في البورصة العالمية", "Business"),
|
| 199 |
+
("علماء يكتشفون علاجاً جديداً يرفع المناعة ضد الفيروسات", "Health"),
|
| 200 |
+
("مفاوضات سياسية جديدة في قمة مجلس الأمن الدولي", "Politics"),
|
| 201 |
+
("سوني تعلن عن ألعاب جديدة لجهاز بلايستيشن", "Gaming"),
|
| 202 |
+
("وكالة ناسا تطلق تلسكوب فضاء جديد لاستكشاف المريخ", "Science"),
|
| 203 |
+
("عرض أول لفيلم سينمائي جديد يتصدر شباك التذاكر", "Entertainment"),
|
| 204 |
+
]
|
| 205 |
+
for text, cat in sample_ar_news * 50:
|
| 206 |
+
all_samples.append({"text": text, "label": cat, "language": "ar"})
|
| 207 |
+
|
| 208 |
+
# --- Shuffle and split 90/10 ---
|
| 209 |
+
import random
|
| 210 |
+
random.shuffle(all_samples)
|
| 211 |
+
split_idx = int(len(all_samples) * 0.9)
|
| 212 |
+
train_data = all_samples[:split_idx]
|
| 213 |
+
test_data = all_samples[split_idx:]
|
| 214 |
+
|
| 215 |
+
_save_jsonl(train_data, os.path.join(output_dir, "category_train.jsonl"))
|
| 216 |
+
_save_jsonl(test_data, os.path.join(output_dir, "category_test.jsonl"))
|
| 217 |
+
logger.info(f"Category dataset saved: {len(train_data)} train, {len(test_data)} test")
|
| 218 |
+
|
| 219 |
+
|
| 220 |
+
def prepare_sentiment_datasets(output_dir: str):
|
| 221 |
+
"""
|
| 222 |
+
Download and prepare Sentiment Analysis datasets (Arabic + English).
|
| 223 |
+
Outputs: sentiment_train.jsonl, sentiment_test.jsonl with {text, label, language} pairs.
|
| 224 |
+
"""
|
| 225 |
+
from datasets import load_dataset
|
| 226 |
+
|
| 227 |
+
os.makedirs(output_dir, exist_ok=True)
|
| 228 |
+
all_samples = []
|
| 229 |
+
|
| 230 |
+
# --- English: SST-2 / TweetEval ---
|
| 231 |
+
logger.info("Loading English sentiment dataset...")
|
| 232 |
+
en_loaded = False
|
| 233 |
+
for candidate in [("stanfordnlp/sst2", None), ("nyu-mll/glue", "sst2"), ("mteb/tweet_eval_sentiment", None)]:
|
| 234 |
+
try:
|
| 235 |
+
ds_name, sub = candidate
|
| 236 |
+
sst2 = load_dataset(ds_name, sub, split="train") if sub else load_dataset(ds_name, split="train")
|
| 237 |
+
en_count = 0
|
| 238 |
+
for sample in sst2:
|
| 239 |
+
text = sample.get("sentence") or sample.get("text") or ""
|
| 240 |
+
label = sample.get("label", 0)
|
| 241 |
+
mapped = "Positive" if label == 1 else "Negative" if label == 0 else "Neutral"
|
| 242 |
+
if text:
|
| 243 |
+
all_samples.append({"text": text[:512], "label": mapped, "language": "en"})
|
| 244 |
+
en_count += 1
|
| 245 |
+
if en_count > 0:
|
| 246 |
+
logger.info(f"Loaded {en_count} English sentiment samples from '{ds_name}'")
|
| 247 |
+
en_loaded = True
|
| 248 |
+
break
|
| 249 |
+
except Exception:
|
| 250 |
+
continue
|
| 251 |
+
|
| 252 |
+
# --- Arabic: TweetEval / Sentiment ---
|
| 253 |
+
logger.info("Loading Arabic sentiment dataset...")
|
| 254 |
+
ar_loaded = False
|
| 255 |
+
for candidate in ["mteb/tweet_eval_sentiment", "MohamedAtef/LABR"]:
|
| 256 |
+
try:
|
| 257 |
+
asad = load_dataset(candidate, split="train")
|
| 258 |
+
ar_count = 0
|
| 259 |
+
for sample in asad:
|
| 260 |
+
text = sample.get("text") or sample.get("text_raw") or ""
|
| 261 |
+
label = sample.get("label", 1)
|
| 262 |
+
label_map = {0: "Negative", 1: "Neutral", 2: "Positive"}
|
| 263 |
+
mapped = label_map.get(label, "Neutral")
|
| 264 |
+
if text:
|
| 265 |
+
all_samples.append({"text": text[:512], "label": mapped, "language": "ar"})
|
| 266 |
+
ar_count += 1
|
| 267 |
+
if ar_count > 0:
|
| 268 |
+
logger.info(f"Loaded {ar_count} Arabic sentiment samples")
|
| 269 |
+
ar_loaded = True
|
| 270 |
+
break
|
| 271 |
+
except Exception:
|
| 272 |
+
continue
|
| 273 |
+
|
| 274 |
+
if not ar_loaded and len([s for s in all_samples if s["language"] == "ar"]) < 10:
|
| 275 |
+
sample_ar_sentiment = [
|
| 276 |
+
("فوز تاريخي وانتعاش اقتصادي ممتاز للمنطقة", "Positive"),
|
| 277 |
+
("تراجع كبير وخسائر ضخمة وتدهور الأوضاع", "Negative"),
|
| 278 |
+
("عقد اجتماع عادي لمناقشة جدول الأعمال", "Neutral"),
|
| 279 |
+
]
|
| 280 |
+
for text, sent in sample_ar_sentiment * 100:
|
| 281 |
+
all_samples.append({"text": text, "label": sent, "language": "ar"})
|
| 282 |
+
|
| 283 |
+
# --- Shuffle and split ---
|
| 284 |
+
import random
|
| 285 |
+
random.shuffle(all_samples)
|
| 286 |
+
split_idx = int(len(all_samples) * 0.9)
|
| 287 |
+
|
| 288 |
+
_save_jsonl(all_samples[:split_idx], os.path.join(output_dir, "sentiment_train.jsonl"))
|
| 289 |
+
_save_jsonl(all_samples[split_idx:], os.path.join(output_dir, "sentiment_test.jsonl"))
|
| 290 |
+
logger.info(f"Sentiment dataset saved: {split_idx} train, {len(all_samples) - split_idx} test")
|
| 291 |
+
|
| 292 |
+
|
| 293 |
+
def prepare_ner_datasets(output_dir: str):
|
| 294 |
+
"""
|
| 295 |
+
Download and prepare WikiANN / CoNLL-2003 for NER.
|
| 296 |
+
Outputs: ner_train.jsonl, ner_test.jsonl with {tokens, ner_tags, language} pairs.
|
| 297 |
+
"""
|
| 298 |
+
from datasets import load_dataset
|
| 299 |
+
|
| 300 |
+
os.makedirs(output_dir, exist_ok=True)
|
| 301 |
+
all_train = []
|
| 302 |
+
all_test = []
|
| 303 |
+
|
| 304 |
+
# --- WikiANN Arabic ---
|
| 305 |
+
logger.info("Loading WikiANN Arabic NER dataset...")
|
| 306 |
+
try:
|
| 307 |
+
wikiann_ar = load_dataset("unimelb-nlp/wikiann", "ar")
|
| 308 |
+
ar_train = [{"tokens": s["tokens"], "ner_tags": s["ner_tags"], "language": "ar"} for s in wikiann_ar["train"]]
|
| 309 |
+
ar_test = [{"tokens": s["tokens"], "ner_tags": s["ner_tags"], "language": "ar"} for s in wikiann_ar["test"]]
|
| 310 |
+
all_train.extend(ar_train)
|
| 311 |
+
all_test.extend(ar_test)
|
| 312 |
+
logger.info(f"Loaded WikiANN Arabic: {len(ar_train)} train, {len(ar_test)} test")
|
| 313 |
+
except Exception as e:
|
| 314 |
+
logger.warning(f"Could not load WikiANN Arabic: {e}")
|
| 315 |
+
|
| 316 |
+
# --- WikiANN English ---
|
| 317 |
+
logger.info("Loading WikiANN English NER dataset...")
|
| 318 |
+
try:
|
| 319 |
+
wikiann_en = load_dataset("unimelb-nlp/wikiann", "en")
|
| 320 |
+
en_train = [{"tokens": s["tokens"], "ner_tags": s["ner_tags"], "language": "en"} for s in wikiann_en["train"]]
|
| 321 |
+
en_test = [{"tokens": s["tokens"], "ner_tags": s["ner_tags"], "language": "en"} for s in wikiann_en["test"]]
|
| 322 |
+
all_train.extend(en_train)
|
| 323 |
+
all_test.extend(en_test)
|
| 324 |
+
logger.info(f"Loaded WikiANN English: {len(en_train)} train, {len(en_test)} test")
|
| 325 |
+
except Exception as e:
|
| 326 |
+
logger.warning(f"Could not load WikiANN English: {e}")
|
| 327 |
+
|
| 328 |
+
# --- CoNLL-2003 English ---
|
| 329 |
+
logger.info("Loading CoNLL-2003 English NER dataset...")
|
| 330 |
+
try:
|
| 331 |
+
conll = load_dataset("eriktks/conll2003")
|
| 332 |
+
conll_train = [{"tokens": s["tokens"], "ner_tags": s["ner_tags"], "language": "en"} for s in conll["train"]]
|
| 333 |
+
conll_test = [{"tokens": s["tokens"], "ner_tags": s["ner_tags"], "language": "en"} for s in conll["test"]]
|
| 334 |
+
all_train.extend(conll_train)
|
| 335 |
+
all_test.extend(conll_test)
|
| 336 |
+
logger.info(f"Added CoNLL-2003: {len(conll_train)} train, {len(conll_test)} test")
|
| 337 |
+
except Exception as e:
|
| 338 |
+
logger.warning(f"Could not load CoNLL-2003: {e}")
|
| 339 |
+
|
| 340 |
+
# Fallback if datasets were unavailable
|
| 341 |
+
if not all_train:
|
| 342 |
+
sample_ner = [
|
| 343 |
+
{"tokens": ["الرئيس", "السيسي", "في", "القاهرة"], "ner_tags": [0, 1, 0, 5], "language": "ar"},
|
| 344 |
+
{"tokens": ["Biden", "visited", "London"], "ner_tags": [1, 0, 5], "language": "en"}
|
| 345 |
+
]
|
| 346 |
+
all_train = sample_ner * 50
|
| 347 |
+
all_test = sample_ner * 10
|
| 348 |
+
|
| 349 |
+
_save_jsonl(all_train, os.path.join(output_dir, "ner_train.jsonl"))
|
| 350 |
+
_save_jsonl(all_test, os.path.join(output_dir, "ner_test.jsonl"))
|
| 351 |
+
logger.info(f"NER dataset saved: {len(all_train)} train, {len(all_test)} test")
|
| 352 |
+
|
| 353 |
+
|
| 354 |
+
def _save_jsonl(data: list, filepath: str):
|
| 355 |
+
"""Save a list of dicts as JSONL, merging with any existing file to preserve collected data."""
|
| 356 |
+
existing_samples = []
|
| 357 |
+
if os.path.exists(filepath):
|
| 358 |
+
try:
|
| 359 |
+
with open(filepath, 'r', encoding='utf-8') as f:
|
| 360 |
+
for line in f:
|
| 361 |
+
if line.strip():
|
| 362 |
+
existing_samples.append(json.loads(line.strip()))
|
| 363 |
+
except Exception:
|
| 364 |
+
pass
|
| 365 |
+
|
| 366 |
+
# Merge and deduplicate by text/tokens content
|
| 367 |
+
seen_keys = set()
|
| 368 |
+
combined_data = []
|
| 369 |
+
|
| 370 |
+
for item in existing_samples + data:
|
| 371 |
+
key = str(item.get("text") or item.get("tokens") or "")
|
| 372 |
+
if key and key not in seen_keys:
|
| 373 |
+
seen_keys.add(key)
|
| 374 |
+
combined_data.append(item)
|
| 375 |
+
|
| 376 |
+
with open(filepath, 'w', encoding='utf-8') as f:
|
| 377 |
+
for item in combined_data:
|
| 378 |
+
f.write(json.dumps(item, ensure_ascii=False) + '\n')
|
| 379 |
+
|
| 380 |
+
new_added = len(combined_data) - len(existing_samples)
|
| 381 |
+
logger.info(f"Saved {len(combined_data)} total samples ({len(existing_samples)} existing + {max(0, new_added)} new) to {filepath}")
|
| 382 |
+
|
| 383 |
+
|
| 384 |
+
if __name__ == "__main__":
|
| 385 |
+
parser = argparse.ArgumentParser(description="Prepare NLP training datasets for WhatsNew")
|
| 386 |
+
parser.add_argument("--task", choices=["category", "sentiment", "ner", "all"], default="all",
|
| 387 |
+
help="Which dataset to prepare")
|
| 388 |
+
parser.add_argument("--output-dir", default="nlp_training/data", help="Output directory for processed datasets")
|
| 389 |
+
args = parser.parse_args()
|
| 390 |
+
|
| 391 |
+
if args.task in ("category", "all"):
|
| 392 |
+
prepare_category_datasets(args.output_dir)
|
| 393 |
+
if args.task in ("sentiment", "all"):
|
| 394 |
+
prepare_sentiment_datasets(args.output_dir)
|
| 395 |
+
if args.task in ("ner", "all"):
|
| 396 |
+
prepare_ner_datasets(args.output_dir)
|
| 397 |
+
|
| 398 |
+
logger.info("Dataset preparation complete!")
|
nlp_training/train_category_classifier.py
ADDED
|
@@ -0,0 +1,190 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Fine-tune a multilingual category classifier for WhatsNew News Intelligence Engine.
|
| 3 |
+
|
| 4 |
+
Model: XLM-RoBERTa-base (handles Arabic + English in a single model)
|
| 5 |
+
Dataset: SANAD (Arabic) + HuffPost (English) → 13 unified news categories
|
| 6 |
+
|
| 7 |
+
Usage:
|
| 8 |
+
# 1. First prepare the data:
|
| 9 |
+
python nlp_training/prepare_datasets.py --task category
|
| 10 |
+
|
| 11 |
+
# 2. Then fine-tune:
|
| 12 |
+
python nlp_training/train_category_classifier.py
|
| 13 |
+
|
| 14 |
+
# 3. The fine-tuned model will be saved to ./models/category_classifier/
|
| 15 |
+
# and can be uploaded to HuggingFace Hub with:
|
| 16 |
+
python nlp_training/train_category_classifier.py --push-to-hub AbdoCrow/whatsnew-category
|
| 17 |
+
|
| 18 |
+
Requirements:
|
| 19 |
+
pip install transformers datasets torch scikit-learn accelerate
|
| 20 |
+
"""
|
| 21 |
+
|
| 22 |
+
import os
|
| 23 |
+
import json
|
| 24 |
+
import logging
|
| 25 |
+
import argparse
|
| 26 |
+
import numpy as np
|
| 27 |
+
|
| 28 |
+
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
|
| 29 |
+
logger = logging.getLogger(__name__)
|
| 30 |
+
|
| 31 |
+
# Our unified 13 categories
|
| 32 |
+
LABEL_LIST = [
|
| 33 |
+
"Sports", "Technology", "Gaming", "Business", "Health", "Science",
|
| 34 |
+
"Food", "Entertainment", "Politics", "Education", "Crime", "Weather", "Travel"
|
| 35 |
+
]
|
| 36 |
+
LABEL2ID = {label: i for i, label in enumerate(LABEL_LIST)}
|
| 37 |
+
ID2LABEL = {i: label for i, label in enumerate(LABEL_LIST)}
|
| 38 |
+
|
| 39 |
+
|
| 40 |
+
def load_jsonl(filepath):
|
| 41 |
+
"""Load a JSONL dataset file."""
|
| 42 |
+
samples = []
|
| 43 |
+
with open(filepath, 'r', encoding='utf-8') as f:
|
| 44 |
+
for line in f:
|
| 45 |
+
samples.append(json.loads(line.strip()))
|
| 46 |
+
return samples
|
| 47 |
+
|
| 48 |
+
|
| 49 |
+
def main(args):
|
| 50 |
+
from transformers import (
|
| 51 |
+
AutoTokenizer,
|
| 52 |
+
AutoModelForSequenceClassification,
|
| 53 |
+
TrainingArguments,
|
| 54 |
+
Trainer,
|
| 55 |
+
EarlyStoppingCallback
|
| 56 |
+
)
|
| 57 |
+
from datasets import Dataset
|
| 58 |
+
from sklearn.metrics import accuracy_score, f1_score
|
| 59 |
+
|
| 60 |
+
# ---------------------------------------------------------------
|
| 61 |
+
# 1. Load prepared datasets
|
| 62 |
+
# ---------------------------------------------------------------
|
| 63 |
+
data_dir = args.data_dir
|
| 64 |
+
train_path = os.path.join(data_dir, "category_train.jsonl")
|
| 65 |
+
test_path = os.path.join(data_dir, "category_test.jsonl")
|
| 66 |
+
|
| 67 |
+
if not os.path.exists(train_path):
|
| 68 |
+
logger.error(f"Training data not found at {train_path}. Run prepare_datasets.py first.")
|
| 69 |
+
return
|
| 70 |
+
|
| 71 |
+
train_data = load_jsonl(train_path)
|
| 72 |
+
test_data = load_jsonl(test_path)
|
| 73 |
+
|
| 74 |
+
# Filter to only include our known labels
|
| 75 |
+
train_data = [s for s in train_data if s["label"] in LABEL2ID]
|
| 76 |
+
test_data = [s for s in test_data if s["label"] in LABEL2ID]
|
| 77 |
+
|
| 78 |
+
logger.info(f"Train: {len(train_data)} samples, Test: {len(test_data)} samples")
|
| 79 |
+
|
| 80 |
+
# Convert to HuggingFace Dataset
|
| 81 |
+
train_dataset = Dataset.from_list([
|
| 82 |
+
{"text": s["text"], "label": LABEL2ID[s["label"]]} for s in train_data
|
| 83 |
+
])
|
| 84 |
+
test_dataset = Dataset.from_list([
|
| 85 |
+
{"text": s["text"], "label": LABEL2ID[s["label"]]} for s in test_data
|
| 86 |
+
])
|
| 87 |
+
|
| 88 |
+
# ---------------------------------------------------------------
|
| 89 |
+
# 2. Tokenize
|
| 90 |
+
# ---------------------------------------------------------------
|
| 91 |
+
model_name = args.model_name
|
| 92 |
+
tokenizer = AutoTokenizer.from_pretrained(model_name)
|
| 93 |
+
|
| 94 |
+
def tokenize_fn(examples):
|
| 95 |
+
return tokenizer(examples["text"], truncation=True, padding="max_length", max_length=256)
|
| 96 |
+
|
| 97 |
+
train_dataset = train_dataset.map(tokenize_fn, batched=True, remove_columns=["text"])
|
| 98 |
+
test_dataset = test_dataset.map(tokenize_fn, batched=True, remove_columns=["text"])
|
| 99 |
+
|
| 100 |
+
train_dataset.set_format("torch")
|
| 101 |
+
test_dataset.set_format("torch")
|
| 102 |
+
|
| 103 |
+
# ---------------------------------------------------------------
|
| 104 |
+
# 3. Load Model
|
| 105 |
+
# ---------------------------------------------------------------
|
| 106 |
+
model = AutoModelForSequenceClassification.from_pretrained(
|
| 107 |
+
model_name,
|
| 108 |
+
num_labels=len(LABEL_LIST),
|
| 109 |
+
id2label=ID2LABEL,
|
| 110 |
+
label2id=LABEL2ID
|
| 111 |
+
)
|
| 112 |
+
logger.info(f"Loaded model: {model_name} with {len(LABEL_LIST)} labels")
|
| 113 |
+
|
| 114 |
+
# ---------------------------------------------------------------
|
| 115 |
+
# 4. Training Configuration
|
| 116 |
+
# ---------------------------------------------------------------
|
| 117 |
+
output_dir = args.output_dir
|
| 118 |
+
|
| 119 |
+
training_args = TrainingArguments(
|
| 120 |
+
output_dir=output_dir,
|
| 121 |
+
num_train_epochs=args.epochs,
|
| 122 |
+
per_device_train_batch_size=args.batch_size,
|
| 123 |
+
per_device_eval_batch_size=args.batch_size * 2,
|
| 124 |
+
learning_rate=args.learning_rate,
|
| 125 |
+
weight_decay=0.01,
|
| 126 |
+
eval_strategy="epoch",
|
| 127 |
+
save_strategy="epoch",
|
| 128 |
+
load_best_model_at_end=True,
|
| 129 |
+
metric_for_best_model="f1_macro",
|
| 130 |
+
greater_is_better=True,
|
| 131 |
+
save_total_limit=2,
|
| 132 |
+
logging_steps=100,
|
| 133 |
+
warmup_ratio=0.1,
|
| 134 |
+
fp16=args.fp16,
|
| 135 |
+
report_to="none", # Disable wandb/tensorboard
|
| 136 |
+
)
|
| 137 |
+
|
| 138 |
+
def compute_metrics(eval_pred):
|
| 139 |
+
logits, labels = eval_pred
|
| 140 |
+
predictions = np.argmax(logits, axis=-1)
|
| 141 |
+
acc = accuracy_score(labels, predictions)
|
| 142 |
+
f1 = f1_score(labels, predictions, average="macro")
|
| 143 |
+
return {"accuracy": acc, "f1_macro": f1}
|
| 144 |
+
|
| 145 |
+
trainer = Trainer(
|
| 146 |
+
model=model,
|
| 147 |
+
args=training_args,
|
| 148 |
+
train_dataset=train_dataset,
|
| 149 |
+
eval_dataset=test_dataset,
|
| 150 |
+
compute_metrics=compute_metrics,
|
| 151 |
+
callbacks=[EarlyStoppingCallback(early_stopping_patience=2)]
|
| 152 |
+
)
|
| 153 |
+
|
| 154 |
+
# ---------------------------------------------------------------
|
| 155 |
+
# 5. Train
|
| 156 |
+
# ---------------------------------------------------------------
|
| 157 |
+
logger.info("Starting training...")
|
| 158 |
+
trainer.train()
|
| 159 |
+
|
| 160 |
+
# ---------------------------------------------------------------
|
| 161 |
+
# 6. Evaluate
|
| 162 |
+
# ---------------------------------------------------------------
|
| 163 |
+
results = trainer.evaluate()
|
| 164 |
+
logger.info(f"Evaluation results: {results}")
|
| 165 |
+
|
| 166 |
+
# ---------------------------------------------------------------
|
| 167 |
+
# 7. Save
|
| 168 |
+
# ---------------------------------------------------------------
|
| 169 |
+
trainer.save_model(output_dir)
|
| 170 |
+
tokenizer.save_pretrained(output_dir)
|
| 171 |
+
logger.info(f"Model saved to {output_dir}")
|
| 172 |
+
|
| 173 |
+
# Optionally push to HuggingFace Hub
|
| 174 |
+
if args.push_to_hub:
|
| 175 |
+
trainer.push_to_hub(args.push_to_hub)
|
| 176 |
+
logger.info(f"Model pushed to HuggingFace Hub: {args.push_to_hub}")
|
| 177 |
+
|
| 178 |
+
|
| 179 |
+
if __name__ == "__main__":
|
| 180 |
+
parser = argparse.ArgumentParser(description="Fine-tune a news category classifier")
|
| 181 |
+
parser.add_argument("--model-name", default="xlm-roberta-base", help="Base model name")
|
| 182 |
+
parser.add_argument("--data-dir", default="nlp_training/data", help="Directory with prepared datasets")
|
| 183 |
+
parser.add_argument("--output-dir", default="models/category_classifier", help="Output directory for trained model")
|
| 184 |
+
parser.add_argument("--epochs", type=int, default=5, help="Number of training epochs")
|
| 185 |
+
parser.add_argument("--batch-size", type=int, default=16, help="Training batch size")
|
| 186 |
+
parser.add_argument("--learning-rate", type=float, default=2e-5, help="Learning rate")
|
| 187 |
+
parser.add_argument("--fp16", action="store_true", help="Use mixed precision training (requires CUDA)")
|
| 188 |
+
parser.add_argument("--push-to-hub", type=str, default=None, help="HuggingFace Hub repo to push to")
|
| 189 |
+
args = parser.parse_args()
|
| 190 |
+
main(args)
|
nlp_training/train_ner_model.py
ADDED
|
@@ -0,0 +1,271 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Fine-tune a multilingual NER model for WhatsNew News Intelligence Engine.
|
| 3 |
+
|
| 4 |
+
Model: XLM-RoBERTa-base
|
| 5 |
+
Dataset: WikiANN (Arabic + English) + CoNLL-2003 (English)
|
| 6 |
+
Entities: PER (people), ORG (organizations), LOC (locations)
|
| 7 |
+
|
| 8 |
+
Usage:
|
| 9 |
+
# 1. First prepare the data:
|
| 10 |
+
python nlp_training/prepare_datasets.py --task ner
|
| 11 |
+
|
| 12 |
+
# 2. Then fine-tune:
|
| 13 |
+
python nlp_training/train_ner_model.py
|
| 14 |
+
|
| 15 |
+
# 3. The fine-tuned model will be saved to ./models/ner_model/
|
| 16 |
+
|
| 17 |
+
Requirements:
|
| 18 |
+
pip install transformers datasets torch scikit-learn seqeval accelerate
|
| 19 |
+
"""
|
| 20 |
+
|
| 21 |
+
import os
|
| 22 |
+
import json
|
| 23 |
+
import logging
|
| 24 |
+
import argparse
|
| 25 |
+
import numpy as np
|
| 26 |
+
|
| 27 |
+
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
|
| 28 |
+
logger = logging.getLogger(__name__)
|
| 29 |
+
|
| 30 |
+
# WikiANN NER tag mapping (IOB2 format)
|
| 31 |
+
# WikiANN uses: O=0, B-PER=1, I-PER=2, B-ORG=3, I-ORG=4, B-LOC=5, I-LOC=6
|
| 32 |
+
LABEL_LIST = ["O", "B-PER", "I-PER", "B-ORG", "I-ORG", "B-LOC", "I-LOC"]
|
| 33 |
+
LABEL2ID = {label: i for i, label in enumerate(LABEL_LIST)}
|
| 34 |
+
ID2LABEL = {i: label for i, label in enumerate(LABEL_LIST)}
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
def load_jsonl(filepath):
|
| 38 |
+
samples = []
|
| 39 |
+
with open(filepath, 'r', encoding='utf-8') as f:
|
| 40 |
+
for line in f:
|
| 41 |
+
samples.append(json.loads(line.strip()))
|
| 42 |
+
return samples
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
def main(args):
|
| 46 |
+
from transformers import (
|
| 47 |
+
AutoTokenizer,
|
| 48 |
+
AutoModelForTokenClassification,
|
| 49 |
+
TrainingArguments,
|
| 50 |
+
Trainer,
|
| 51 |
+
DataCollatorForTokenClassification,
|
| 52 |
+
EarlyStoppingCallback
|
| 53 |
+
)
|
| 54 |
+
from datasets import Dataset
|
| 55 |
+
|
| 56 |
+
# ---------------------------------------------------------------
|
| 57 |
+
# 1. Load prepared datasets
|
| 58 |
+
# ---------------------------------------------------------------
|
| 59 |
+
data_dir = args.data_dir
|
| 60 |
+
train_path = os.path.join(data_dir, "ner_train.jsonl")
|
| 61 |
+
test_path = os.path.join(data_dir, "ner_test.jsonl")
|
| 62 |
+
|
| 63 |
+
if not os.path.exists(train_path):
|
| 64 |
+
logger.error(f"Training data not found at {train_path}. Run prepare_datasets.py first.")
|
| 65 |
+
return
|
| 66 |
+
|
| 67 |
+
train_data = load_jsonl(train_path)
|
| 68 |
+
test_data = load_jsonl(test_path)
|
| 69 |
+
|
| 70 |
+
logger.info(f"Train: {len(train_data)} samples, Test: {len(test_data)} samples")
|
| 71 |
+
|
| 72 |
+
train_dataset = Dataset.from_list(train_data)
|
| 73 |
+
test_dataset = Dataset.from_list(test_data)
|
| 74 |
+
|
| 75 |
+
# ---------------------------------------------------------------
|
| 76 |
+
# 2. Tokenize with label alignment
|
| 77 |
+
# ---------------------------------------------------------------
|
| 78 |
+
model_name = args.model_name
|
| 79 |
+
tokenizer = AutoTokenizer.from_pretrained(model_name)
|
| 80 |
+
|
| 81 |
+
def tokenize_and_align_labels(examples):
|
| 82 |
+
"""
|
| 83 |
+
Tokenize text and realign NER labels with sub-word tokens.
|
| 84 |
+
Sub-word continuations receive label -100 (ignored by loss function).
|
| 85 |
+
"""
|
| 86 |
+
tokenized_inputs = tokenizer(
|
| 87 |
+
examples["tokens"],
|
| 88 |
+
truncation=True,
|
| 89 |
+
padding="max_length",
|
| 90 |
+
max_length=128,
|
| 91 |
+
is_split_into_words=True
|
| 92 |
+
)
|
| 93 |
+
|
| 94 |
+
all_labels = []
|
| 95 |
+
for i, labels in enumerate(examples["ner_tags"]):
|
| 96 |
+
word_ids = tokenized_inputs.word_ids(batch_index=i)
|
| 97 |
+
label_ids = []
|
| 98 |
+
previous_word_idx = None
|
| 99 |
+
for word_idx in word_ids:
|
| 100 |
+
if word_idx is None:
|
| 101 |
+
label_ids.append(-100) # Special tokens
|
| 102 |
+
elif word_idx != previous_word_idx:
|
| 103 |
+
# Map WikiANN integer tags to our label space
|
| 104 |
+
tag = labels[word_idx] if word_idx < len(labels) else 0
|
| 105 |
+
label_ids.append(tag)
|
| 106 |
+
else:
|
| 107 |
+
# Sub-word continuation: use -100 or keep I- tag
|
| 108 |
+
tag = labels[word_idx] if word_idx < len(labels) else 0
|
| 109 |
+
# For sub-words of an entity, keep the I- variant
|
| 110 |
+
if tag in (1, 3, 5): # B-PER, B-ORG, B-LOC → convert to I-
|
| 111 |
+
tag = tag + 1
|
| 112 |
+
label_ids.append(tag)
|
| 113 |
+
previous_word_idx = word_idx
|
| 114 |
+
all_labels.append(label_ids)
|
| 115 |
+
|
| 116 |
+
tokenized_inputs["labels"] = all_labels
|
| 117 |
+
return tokenized_inputs
|
| 118 |
+
|
| 119 |
+
# Remove the "language" column if present
|
| 120 |
+
columns_to_remove = ["tokens", "ner_tags"]
|
| 121 |
+
if "language" in train_dataset.column_names:
|
| 122 |
+
columns_to_remove.append("language")
|
| 123 |
+
|
| 124 |
+
train_dataset = train_dataset.map(
|
| 125 |
+
tokenize_and_align_labels, batched=True, remove_columns=columns_to_remove
|
| 126 |
+
)
|
| 127 |
+
test_dataset = test_dataset.map(
|
| 128 |
+
tokenize_and_align_labels, batched=True, remove_columns=columns_to_remove
|
| 129 |
+
)
|
| 130 |
+
|
| 131 |
+
train_dataset.set_format("torch")
|
| 132 |
+
test_dataset.set_format("torch")
|
| 133 |
+
|
| 134 |
+
# ---------------------------------------------------------------
|
| 135 |
+
# 3. Load Model
|
| 136 |
+
# ---------------------------------------------------------------
|
| 137 |
+
model = AutoModelForTokenClassification.from_pretrained(
|
| 138 |
+
model_name,
|
| 139 |
+
num_labels=len(LABEL_LIST),
|
| 140 |
+
id2label=ID2LABEL,
|
| 141 |
+
label2id=LABEL2ID
|
| 142 |
+
)
|
| 143 |
+
logger.info(f"Loaded model: {model_name}")
|
| 144 |
+
|
| 145 |
+
# ---------------------------------------------------------------
|
| 146 |
+
# 4. Metrics (using seqeval for entity-level F1)
|
| 147 |
+
# ---------------------------------------------------------------
|
| 148 |
+
try:
|
| 149 |
+
from seqeval.metrics import f1_score as seqeval_f1, classification_report as seqeval_report
|
| 150 |
+
HAS_SEQEVAL = True
|
| 151 |
+
except ImportError:
|
| 152 |
+
logger.warning("seqeval not installed. Using token-level accuracy only.")
|
| 153 |
+
HAS_SEQEVAL = False
|
| 154 |
+
|
| 155 |
+
def compute_metrics(eval_pred):
|
| 156 |
+
logits, labels = eval_pred
|
| 157 |
+
predictions = np.argmax(logits, axis=-1)
|
| 158 |
+
|
| 159 |
+
if HAS_SEQEVAL:
|
| 160 |
+
# Convert to label strings, skipping -100
|
| 161 |
+
true_labels_seq = []
|
| 162 |
+
pred_labels_seq = []
|
| 163 |
+
for pred_seq, label_seq in zip(predictions, labels):
|
| 164 |
+
true_sent = []
|
| 165 |
+
pred_sent = []
|
| 166 |
+
for p, l in zip(pred_seq, label_seq):
|
| 167 |
+
if l != -100:
|
| 168 |
+
true_sent.append(ID2LABEL.get(int(l), "O"))
|
| 169 |
+
pred_sent.append(ID2LABEL.get(int(p), "O"))
|
| 170 |
+
true_labels_seq.append(true_sent)
|
| 171 |
+
pred_labels_seq.append(pred_sent)
|
| 172 |
+
|
| 173 |
+
f1 = seqeval_f1(true_labels_seq, pred_labels_seq, average="macro")
|
| 174 |
+
return {"f1_macro": f1}
|
| 175 |
+
else:
|
| 176 |
+
# Fallback: token-level accuracy
|
| 177 |
+
mask = labels != -100
|
| 178 |
+
correct = (predictions[mask] == labels[mask]).sum()
|
| 179 |
+
total = mask.sum()
|
| 180 |
+
return {"accuracy": float(correct / total)}
|
| 181 |
+
|
| 182 |
+
# ---------------------------------------------------------------
|
| 183 |
+
# 5. Training Configuration
|
| 184 |
+
# ---------------------------------------------------------------
|
| 185 |
+
output_dir = args.output_dir
|
| 186 |
+
data_collator = DataCollatorForTokenClassification(tokenizer)
|
| 187 |
+
|
| 188 |
+
training_args = TrainingArguments(
|
| 189 |
+
output_dir=output_dir,
|
| 190 |
+
num_train_epochs=args.epochs,
|
| 191 |
+
per_device_train_batch_size=args.batch_size,
|
| 192 |
+
per_device_eval_batch_size=args.batch_size * 2,
|
| 193 |
+
learning_rate=args.learning_rate,
|
| 194 |
+
weight_decay=0.01,
|
| 195 |
+
eval_strategy="epoch",
|
| 196 |
+
save_strategy="epoch",
|
| 197 |
+
load_best_model_at_end=True,
|
| 198 |
+
metric_for_best_model="f1_macro" if HAS_SEQEVAL else "accuracy",
|
| 199 |
+
greater_is_better=True,
|
| 200 |
+
save_total_limit=2,
|
| 201 |
+
logging_steps=100,
|
| 202 |
+
warmup_ratio=0.1,
|
| 203 |
+
fp16=args.fp16,
|
| 204 |
+
report_to="none",
|
| 205 |
+
)
|
| 206 |
+
|
| 207 |
+
trainer = Trainer(
|
| 208 |
+
model=model,
|
| 209 |
+
args=training_args,
|
| 210 |
+
train_dataset=train_dataset,
|
| 211 |
+
eval_dataset=test_dataset,
|
| 212 |
+
data_collator=data_collator,
|
| 213 |
+
compute_metrics=compute_metrics,
|
| 214 |
+
callbacks=[EarlyStoppingCallback(early_stopping_patience=2)]
|
| 215 |
+
)
|
| 216 |
+
|
| 217 |
+
# ---------------------------------------------------------------
|
| 218 |
+
# 6. Train
|
| 219 |
+
# ---------------------------------------------------------------
|
| 220 |
+
logger.info("Starting NER training...")
|
| 221 |
+
trainer.train()
|
| 222 |
+
|
| 223 |
+
# ---------------------------------------------------------------
|
| 224 |
+
# 7. Evaluate
|
| 225 |
+
# ---------------------------------------------------------------
|
| 226 |
+
results = trainer.evaluate()
|
| 227 |
+
logger.info(f"Evaluation results: {results}")
|
| 228 |
+
|
| 229 |
+
if HAS_SEQEVAL:
|
| 230 |
+
predictions = trainer.predict(test_dataset)
|
| 231 |
+
pred_labels = np.argmax(predictions.predictions, axis=-1)
|
| 232 |
+
true_labels = predictions.label_ids
|
| 233 |
+
|
| 234 |
+
true_labels_seq = []
|
| 235 |
+
pred_labels_seq = []
|
| 236 |
+
for pred_seq, label_seq in zip(pred_labels, true_labels):
|
| 237 |
+
true_sent = []
|
| 238 |
+
pred_sent = []
|
| 239 |
+
for p, l in zip(pred_seq, label_seq):
|
| 240 |
+
if l != -100:
|
| 241 |
+
true_sent.append(ID2LABEL.get(int(l), "O"))
|
| 242 |
+
pred_sent.append(ID2LABEL.get(int(p), "O"))
|
| 243 |
+
true_labels_seq.append(true_sent)
|
| 244 |
+
pred_labels_seq.append(pred_sent)
|
| 245 |
+
|
| 246 |
+
report = seqeval_report(true_labels_seq, pred_labels_seq)
|
| 247 |
+
logger.info(f"\nEntity-level Classification Report:\n{report}")
|
| 248 |
+
|
| 249 |
+
# ---------------------------------------------------------------
|
| 250 |
+
# 8. Save
|
| 251 |
+
# ---------------------------------------------------------------
|
| 252 |
+
trainer.save_model(output_dir)
|
| 253 |
+
tokenizer.save_pretrained(output_dir)
|
| 254 |
+
logger.info(f"Model saved to {output_dir}")
|
| 255 |
+
|
| 256 |
+
if args.push_to_hub:
|
| 257 |
+
trainer.push_to_hub(args.push_to_hub)
|
| 258 |
+
|
| 259 |
+
|
| 260 |
+
if __name__ == "__main__":
|
| 261 |
+
parser = argparse.ArgumentParser(description="Fine-tune a NER model")
|
| 262 |
+
parser.add_argument("--model-name", default="xlm-roberta-base", help="Base model name")
|
| 263 |
+
parser.add_argument("--data-dir", default="nlp_training/data", help="Directory with prepared datasets")
|
| 264 |
+
parser.add_argument("--output-dir", default="models/ner_model", help="Output directory")
|
| 265 |
+
parser.add_argument("--epochs", type=int, default=5, help="Number of training epochs")
|
| 266 |
+
parser.add_argument("--batch-size", type=int, default=16, help="Training batch size")
|
| 267 |
+
parser.add_argument("--learning-rate", type=float, default=3e-5, help="Learning rate")
|
| 268 |
+
parser.add_argument("--fp16", action="store_true", help="Use mixed precision training")
|
| 269 |
+
parser.add_argument("--push-to-hub", type=str, default=None, help="HuggingFace Hub repo")
|
| 270 |
+
args = parser.parse_args()
|
| 271 |
+
main(args)
|
nlp_training/train_sentiment_analyzer.py
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Fine-tune a multilingual sentiment analyzer for WhatsNew News Intelligence Engine.
|
| 3 |
+
|
| 4 |
+
Model: XLM-RoBERTa-base (handles Arabic + English in a single model)
|
| 5 |
+
Dataset: ASAD (Arabic tweets) + SST-2 (English sentences) → 3 sentiment labels
|
| 6 |
+
|
| 7 |
+
Usage:
|
| 8 |
+
# 1. First prepare the data:
|
| 9 |
+
python nlp_training/prepare_datasets.py --task sentiment
|
| 10 |
+
|
| 11 |
+
# 2. Then fine-tune:
|
| 12 |
+
python nlp_training/train_sentiment_analyzer.py
|
| 13 |
+
|
| 14 |
+
# 3. The fine-tuned model will be saved to ./models/sentiment_analyzer/
|
| 15 |
+
|
| 16 |
+
Requirements:
|
| 17 |
+
pip install transformers datasets torch scikit-learn accelerate
|
| 18 |
+
"""
|
| 19 |
+
|
| 20 |
+
import os
|
| 21 |
+
import json
|
| 22 |
+
import logging
|
| 23 |
+
import argparse
|
| 24 |
+
import numpy as np
|
| 25 |
+
|
| 26 |
+
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
|
| 27 |
+
logger = logging.getLogger(__name__)
|
| 28 |
+
|
| 29 |
+
# Sentiment labels
|
| 30 |
+
LABEL_LIST = ["Positive", "Negative", "Neutral"]
|
| 31 |
+
LABEL2ID = {label: i for i, label in enumerate(LABEL_LIST)}
|
| 32 |
+
ID2LABEL = {i: label for i, label in enumerate(LABEL_LIST)}
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
def load_jsonl(filepath):
|
| 36 |
+
samples = []
|
| 37 |
+
with open(filepath, 'r', encoding='utf-8') as f:
|
| 38 |
+
for line in f:
|
| 39 |
+
samples.append(json.loads(line.strip()))
|
| 40 |
+
return samples
|
| 41 |
+
|
| 42 |
+
|
| 43 |
+
def main(args):
|
| 44 |
+
from transformers import (
|
| 45 |
+
AutoTokenizer,
|
| 46 |
+
AutoModelForSequenceClassification,
|
| 47 |
+
TrainingArguments,
|
| 48 |
+
Trainer,
|
| 49 |
+
EarlyStoppingCallback
|
| 50 |
+
)
|
| 51 |
+
from datasets import Dataset
|
| 52 |
+
from sklearn.metrics import accuracy_score, f1_score, classification_report
|
| 53 |
+
|
| 54 |
+
# ---------------------------------------------------------------
|
| 55 |
+
# 1. Load prepared datasets
|
| 56 |
+
# ---------------------------------------------------------------
|
| 57 |
+
data_dir = args.data_dir
|
| 58 |
+
train_path = os.path.join(data_dir, "sentiment_train.jsonl")
|
| 59 |
+
test_path = os.path.join(data_dir, "sentiment_test.jsonl")
|
| 60 |
+
|
| 61 |
+
if not os.path.exists(train_path):
|
| 62 |
+
logger.error(f"Training data not found at {train_path}. Run prepare_datasets.py first.")
|
| 63 |
+
return
|
| 64 |
+
|
| 65 |
+
train_data = load_jsonl(train_path)
|
| 66 |
+
test_data = load_jsonl(test_path)
|
| 67 |
+
|
| 68 |
+
# Filter to only include our known labels
|
| 69 |
+
train_data = [s for s in train_data if s["label"] in LABEL2ID]
|
| 70 |
+
test_data = [s for s in test_data if s["label"] in LABEL2ID]
|
| 71 |
+
|
| 72 |
+
logger.info(f"Train: {len(train_data)} samples, Test: {len(test_data)} samples")
|
| 73 |
+
|
| 74 |
+
# Label distribution
|
| 75 |
+
from collections import Counter
|
| 76 |
+
train_dist = Counter(s["label"] for s in train_data)
|
| 77 |
+
logger.info(f"Train label distribution: {dict(train_dist)}")
|
| 78 |
+
|
| 79 |
+
# Convert to HuggingFace Dataset
|
| 80 |
+
train_dataset = Dataset.from_list([
|
| 81 |
+
{"text": s["text"], "label": LABEL2ID[s["label"]]} for s in train_data
|
| 82 |
+
])
|
| 83 |
+
test_dataset = Dataset.from_list([
|
| 84 |
+
{"text": s["text"], "label": LABEL2ID[s["label"]]} for s in test_data
|
| 85 |
+
])
|
| 86 |
+
|
| 87 |
+
# ---------------------------------------------------------------
|
| 88 |
+
# 2. Tokenize
|
| 89 |
+
# ---------------------------------------------------------------
|
| 90 |
+
model_name = args.model_name
|
| 91 |
+
tokenizer = AutoTokenizer.from_pretrained(model_name)
|
| 92 |
+
|
| 93 |
+
def tokenize_fn(examples):
|
| 94 |
+
return tokenizer(examples["text"], truncation=True, padding="max_length", max_length=128)
|
| 95 |
+
|
| 96 |
+
train_dataset = train_dataset.map(tokenize_fn, batched=True, remove_columns=["text"])
|
| 97 |
+
test_dataset = test_dataset.map(tokenize_fn, batched=True, remove_columns=["text"])
|
| 98 |
+
|
| 99 |
+
train_dataset.set_format("torch")
|
| 100 |
+
test_dataset.set_format("torch")
|
| 101 |
+
|
| 102 |
+
# ---------------------------------------------------------------
|
| 103 |
+
# 3. Load Model
|
| 104 |
+
# ---------------------------------------------------------------
|
| 105 |
+
model = AutoModelForSequenceClassification.from_pretrained(
|
| 106 |
+
model_name,
|
| 107 |
+
num_labels=len(LABEL_LIST),
|
| 108 |
+
id2label=ID2LABEL,
|
| 109 |
+
label2id=LABEL2ID
|
| 110 |
+
)
|
| 111 |
+
logger.info(f"Loaded model: {model_name}")
|
| 112 |
+
|
| 113 |
+
# ---------------------------------------------------------------
|
| 114 |
+
# 4. Training Configuration
|
| 115 |
+
# ---------------------------------------------------------------
|
| 116 |
+
output_dir = args.output_dir
|
| 117 |
+
|
| 118 |
+
training_args = TrainingArguments(
|
| 119 |
+
output_dir=output_dir,
|
| 120 |
+
num_train_epochs=args.epochs,
|
| 121 |
+
per_device_train_batch_size=args.batch_size,
|
| 122 |
+
per_device_eval_batch_size=args.batch_size * 2,
|
| 123 |
+
learning_rate=args.learning_rate,
|
| 124 |
+
weight_decay=0.01,
|
| 125 |
+
eval_strategy="epoch",
|
| 126 |
+
save_strategy="epoch",
|
| 127 |
+
load_best_model_at_end=True,
|
| 128 |
+
metric_for_best_model="f1_macro",
|
| 129 |
+
greater_is_better=True,
|
| 130 |
+
save_total_limit=2,
|
| 131 |
+
logging_steps=100,
|
| 132 |
+
warmup_ratio=0.1,
|
| 133 |
+
fp16=args.fp16,
|
| 134 |
+
report_to="none",
|
| 135 |
+
)
|
| 136 |
+
|
| 137 |
+
def compute_metrics(eval_pred):
|
| 138 |
+
logits, labels = eval_pred
|
| 139 |
+
predictions = np.argmax(logits, axis=-1)
|
| 140 |
+
acc = accuracy_score(labels, predictions)
|
| 141 |
+
f1 = f1_score(labels, predictions, average="macro")
|
| 142 |
+
return {"accuracy": acc, "f1_macro": f1}
|
| 143 |
+
|
| 144 |
+
trainer = Trainer(
|
| 145 |
+
model=model,
|
| 146 |
+
args=training_args,
|
| 147 |
+
train_dataset=train_dataset,
|
| 148 |
+
eval_dataset=test_dataset,
|
| 149 |
+
compute_metrics=compute_metrics,
|
| 150 |
+
callbacks=[EarlyStoppingCallback(early_stopping_patience=2)]
|
| 151 |
+
)
|
| 152 |
+
|
| 153 |
+
# ---------------------------------------------------------------
|
| 154 |
+
# 5. Train
|
| 155 |
+
# ---------------------------------------------------------------
|
| 156 |
+
logger.info("Starting sentiment training...")
|
| 157 |
+
trainer.train()
|
| 158 |
+
|
| 159 |
+
# ---------------------------------------------------------------
|
| 160 |
+
# 6. Evaluate with detailed report
|
| 161 |
+
# ---------------------------------------------------------------
|
| 162 |
+
results = trainer.evaluate()
|
| 163 |
+
logger.info(f"Evaluation results: {results}")
|
| 164 |
+
|
| 165 |
+
# Detailed per-class report
|
| 166 |
+
predictions = trainer.predict(test_dataset)
|
| 167 |
+
pred_labels = np.argmax(predictions.predictions, axis=-1)
|
| 168 |
+
true_labels = predictions.label_ids
|
| 169 |
+
report = classification_report(true_labels, pred_labels, target_names=LABEL_LIST)
|
| 170 |
+
logger.info(f"\nClassification Report:\n{report}")
|
| 171 |
+
|
| 172 |
+
# ---------------------------------------------------------------
|
| 173 |
+
# 7. Save
|
| 174 |
+
# ---------------------------------------------------------------
|
| 175 |
+
trainer.save_model(output_dir)
|
| 176 |
+
tokenizer.save_pretrained(output_dir)
|
| 177 |
+
logger.info(f"Model saved to {output_dir}")
|
| 178 |
+
|
| 179 |
+
if args.push_to_hub:
|
| 180 |
+
trainer.push_to_hub(args.push_to_hub)
|
| 181 |
+
logger.info(f"Model pushed to HuggingFace Hub: {args.push_to_hub}")
|
| 182 |
+
|
| 183 |
+
|
| 184 |
+
if __name__ == "__main__":
|
| 185 |
+
parser = argparse.ArgumentParser(description="Fine-tune a sentiment analyzer")
|
| 186 |
+
parser.add_argument("--model-name", default="xlm-roberta-base", help="Base model name")
|
| 187 |
+
parser.add_argument("--data-dir", default="nlp_training/data", help="Directory with prepared datasets")
|
| 188 |
+
parser.add_argument("--output-dir", default="models/sentiment_analyzer", help="Output directory")
|
| 189 |
+
parser.add_argument("--epochs", type=int, default=4, help="Number of training epochs")
|
| 190 |
+
parser.add_argument("--batch-size", type=int, default=32, help="Training batch size")
|
| 191 |
+
parser.add_argument("--learning-rate", type=float, default=2e-5, help="Learning rate")
|
| 192 |
+
parser.add_argument("--fp16", action="store_true", help="Use mixed precision training")
|
| 193 |
+
parser.add_argument("--push-to-hub", type=str, default=None, help="HuggingFace Hub repo")
|
| 194 |
+
args = parser.parse_args()
|
| 195 |
+
main(args)
|
requirements.txt
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
annotated-types==0.8.0
|
| 2 |
+
anyio==4.14.2
|
| 3 |
+
beautifulsoup4==4.15.0
|
| 4 |
+
CacheControl==0.14.4
|
| 5 |
+
certifi==2026.7.22
|
| 6 |
+
cffi==2.1.1
|
| 7 |
+
charset-normalizer==3.4.9
|
| 8 |
+
cryptography==50.0.0
|
| 9 |
+
feedparser==6.0.14
|
| 10 |
+
feedparser-sgmllib==2.1.0
|
| 11 |
+
firebase_admin==7.5.0
|
| 12 |
+
google-api-core==2.33.0
|
| 13 |
+
google-auth==2.56.2
|
| 14 |
+
google-cloud-core==2.6.0
|
| 15 |
+
google-cloud-firestore==2.28.0
|
| 16 |
+
google-cloud-storage==3.13.0
|
| 17 |
+
google-crc32c==1.8.0
|
| 18 |
+
google-resumable-media==2.10.0
|
| 19 |
+
googleapis-common-protos==1.75.0
|
| 20 |
+
grpcio==1.83.0
|
| 21 |
+
grpcio-status==1.83.0
|
| 22 |
+
h11==0.16.0
|
| 23 |
+
h2==4.4.1
|
| 24 |
+
hpack==4.2.0
|
| 25 |
+
httpcore==1.0.9
|
| 26 |
+
httpx==0.28.1
|
| 27 |
+
hyperframe==6.1.0
|
| 28 |
+
idna==3.18
|
| 29 |
+
msgpack==1.2.1
|
| 30 |
+
proto-plus==1.28.2
|
| 31 |
+
protobuf==7.35.1
|
| 32 |
+
pyasn1==0.6.4
|
| 33 |
+
pyasn1_modules==0.4.2
|
| 34 |
+
pycparser==3.0
|
| 35 |
+
pydantic==2.13.4
|
| 36 |
+
pydantic_core==2.46.4
|
| 37 |
+
PyJWT==2.13.0
|
| 38 |
+
requests==2.34.2
|
| 39 |
+
schedule==1.2.2
|
| 40 |
+
soupsieve==2.9.1
|
| 41 |
+
typing-inspection==0.4.2
|
| 42 |
+
typing_extensions==4.16.0
|
| 43 |
+
urllib3==2.7.0
|
| 44 |
+
|
| 45 |
+
# NLP & Machine Learning Dependencies
|
| 46 |
+
transformers
|
| 47 |
+
datasets
|
| 48 |
+
scikit-learn
|
| 49 |
+
accelerate
|
| 50 |
+
seqeval
|
| 51 |
+
keybert
|
| 52 |
+
sentence-transformers
|
| 53 |
+
numpy
|
| 54 |
+
gradio
|
requirements_space.txt
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
gradio>=4.0.0
|
| 2 |
+
transformers
|
| 3 |
+
torch
|
| 4 |
+
datasets
|
| 5 |
+
keybert
|
| 6 |
+
sentence-transformers
|
| 7 |
+
beautifulsoup4
|
| 8 |
+
requests
|
| 9 |
+
pydantic
|
scrapers/__pycache__/rss_scraper.cpython-314.pyc
ADDED
|
Binary file (7.82 kB). View file
|
|
|
scrapers/base_scraper.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import requests
|
| 2 |
+
from bs4 import BeautifulSoup
|
| 3 |
+
import logging
|
| 4 |
+
from typing import Optional
|
| 5 |
+
|
| 6 |
+
logger = logging.getLogger(__name__)
|
| 7 |
+
|
| 8 |
+
class BaseScraper:
|
| 9 |
+
headers = {
|
| 10 |
+
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/114.0.0.0 Safari/537.36'
|
| 11 |
+
}
|
| 12 |
+
|
| 13 |
+
def fetch_html(self, url: str) -> Optional[BeautifulSoup]:
|
| 14 |
+
"""Fetches the HTML content of the given URL and returns a BeautifulSoup object."""
|
| 15 |
+
try:
|
| 16 |
+
response = requests.get(url, headers=self.headers, timeout=10)
|
| 17 |
+
response.raise_for_status()
|
| 18 |
+
return BeautifulSoup(response.content, 'html.parser')
|
| 19 |
+
except Exception as e:
|
| 20 |
+
logger.error(f"Failed to fetch {url}: {e}")
|
| 21 |
+
return None
|
scrapers/rss_scraper.py
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import datetime
|
| 2 |
+
import logging
|
| 3 |
+
import feedparser
|
| 4 |
+
from typing import List, Optional
|
| 5 |
+
from bs4 import BeautifulSoup
|
| 6 |
+
from domain.entities.article import Article
|
| 7 |
+
from domain.interfaces import ScraperInterface
|
| 8 |
+
import requests
|
| 9 |
+
import time
|
| 10 |
+
|
| 11 |
+
# Set a custom user agent for feedparser to bypass basic bot protections
|
| 12 |
+
feedparser.USER_AGENT = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/115.0.0.0 Safari/537.36"
|
| 13 |
+
|
| 14 |
+
logger = logging.getLogger(__name__)
|
| 15 |
+
|
| 16 |
+
class RssScraper(ScraperInterface):
|
| 17 |
+
def __init__(self, source_name: str, language: str, feed_url: str, category: str = "World"):
|
| 18 |
+
self.source_name = source_name
|
| 19 |
+
self.language = language
|
| 20 |
+
self.feed_url = feed_url
|
| 21 |
+
self.category = category
|
| 22 |
+
|
| 23 |
+
def scrape(self) -> List[Article]:
|
| 24 |
+
articles = []
|
| 25 |
+
try:
|
| 26 |
+
feed = feedparser.parse(self.feed_url)
|
| 27 |
+
|
| 28 |
+
# Check for network or parsing errors that feedparser hides
|
| 29 |
+
if getattr(feed, 'bozo', 0) == 1 and hasattr(feed, 'bozo_exception'):
|
| 30 |
+
logger.warning(f"Feed error for {self.feed_url}: {feed.bozo_exception}")
|
| 31 |
+
if not feed.entries:
|
| 32 |
+
return articles
|
| 33 |
+
|
| 34 |
+
for entry in feed.entries[:10]: # Limit to top 10 per feed
|
| 35 |
+
title = entry.get("title", "")
|
| 36 |
+
link = entry.get("link", "")
|
| 37 |
+
|
| 38 |
+
# Get summary, remove HTML tags
|
| 39 |
+
summary_raw = entry.get("description", "")
|
| 40 |
+
summary = BeautifulSoup(summary_raw, "html.parser").get_text()[:200]
|
| 41 |
+
|
| 42 |
+
if not title or not link:
|
| 43 |
+
continue
|
| 44 |
+
|
| 45 |
+
# Try to extract image (from RSS or OpenGraph meta tags)
|
| 46 |
+
image_url = self._extract_image(entry, link=link)
|
| 47 |
+
|
| 48 |
+
# Parse date or default to now
|
| 49 |
+
try:
|
| 50 |
+
# feedparser parses date into struct_time
|
| 51 |
+
published_at = datetime.datetime.fromtimestamp(time.mktime(entry.published_parsed)) if hasattr(entry, 'published_parsed') and entry.published_parsed else datetime.datetime.now()
|
| 52 |
+
except Exception:
|
| 53 |
+
published_at = datetime.datetime.now()
|
| 54 |
+
|
| 55 |
+
article = Article(
|
| 56 |
+
id=Article.generate_id_from_url(link),
|
| 57 |
+
title=title,
|
| 58 |
+
source=self.source_name,
|
| 59 |
+
url=link,
|
| 60 |
+
imageUrl=image_url,
|
| 61 |
+
hasImage=True if image_url else False,
|
| 62 |
+
summary=summary + "..." if len(summary) >= 200 else summary,
|
| 63 |
+
publishedAt=published_at,
|
| 64 |
+
language=self.language,
|
| 65 |
+
category=self.category
|
| 66 |
+
)
|
| 67 |
+
articles.append(article)
|
| 68 |
+
except Exception as e:
|
| 69 |
+
logger.error(f"Failed to scrape RSS {self.feed_url}: {e}")
|
| 70 |
+
|
| 71 |
+
return articles
|
| 72 |
+
|
| 73 |
+
def _extract_image(self, entry, link: str = None) -> Optional[str]:
|
| 74 |
+
# 1. Check media_content
|
| 75 |
+
if hasattr(entry, 'media_content') and entry.media_content:
|
| 76 |
+
for media in entry.media_content:
|
| 77 |
+
if media.get('url'):
|
| 78 |
+
return media.get('url')
|
| 79 |
+
|
| 80 |
+
# 2. Check media_thumbnail
|
| 81 |
+
if hasattr(entry, 'media_thumbnail') and entry.media_thumbnail:
|
| 82 |
+
return entry.media_thumbnail[0].get('url')
|
| 83 |
+
|
| 84 |
+
# 3. Check enclosures
|
| 85 |
+
if hasattr(entry, 'enclosures') and entry.enclosures:
|
| 86 |
+
for enc in entry.enclosures:
|
| 87 |
+
if 'image' in enc.get('type', '') and enc.get('href'):
|
| 88 |
+
return enc.get('href')
|
| 89 |
+
|
| 90 |
+
# 4. Check links for image type
|
| 91 |
+
if hasattr(entry, 'links'):
|
| 92 |
+
for l in entry.links:
|
| 93 |
+
if 'image' in l.get('type', ''):
|
| 94 |
+
return l.get('href')
|
| 95 |
+
|
| 96 |
+
# 5. Check description/content HTML for <img> tags
|
| 97 |
+
for field in ['description', 'content']:
|
| 98 |
+
if hasattr(entry, field):
|
| 99 |
+
content_list = getattr(entry, field)
|
| 100 |
+
raw_html = content_list[0].get('value', '') if isinstance(content_list, list) else content_list
|
| 101 |
+
soup = BeautifulSoup(str(raw_html), "html.parser")
|
| 102 |
+
img = soup.find('img')
|
| 103 |
+
if img and img.get('src'):
|
| 104 |
+
return img.get('src')
|
| 105 |
+
|
| 106 |
+
# 6. Layer 2: OpenGraph Web Scraping Fallback (Fast 2s timeout)
|
| 107 |
+
if link and link.startswith('http'):
|
| 108 |
+
try:
|
| 109 |
+
headers = {"User-Agent": feedparser.USER_AGENT}
|
| 110 |
+
resp = requests.get(link, headers=headers, timeout=2.5)
|
| 111 |
+
if resp.status_code == 200:
|
| 112 |
+
soup = BeautifulSoup(resp.text, "html.parser")
|
| 113 |
+
# Check og:image, twitter:image, image_src
|
| 114 |
+
og_img = soup.find('meta', property='og:image') or soup.find('meta', attrs={'name': 'og:image'})
|
| 115 |
+
if og_img and og_img.get('content'):
|
| 116 |
+
return og_img.get('content')
|
| 117 |
+
|
| 118 |
+
tw_img = soup.find('meta', attrs={'name': 'twitter:image'}) or soup.find('meta', property='twitter:image')
|
| 119 |
+
if tw_img and tw_img.get('content'):
|
| 120 |
+
return tw_img.get('content')
|
| 121 |
+
|
| 122 |
+
rel_img = soup.find('link', rel='image_src')
|
| 123 |
+
if rel_img and rel_img.get('href'):
|
| 124 |
+
return rel_img.get('href')
|
| 125 |
+
except Exception:
|
| 126 |
+
pass
|
| 127 |
+
|
| 128 |
+
return None
|
services/__pycache__/ai_classifier.cpython-314.pyc
ADDED
|
Binary file (11.7 kB). View file
|
|
|
services/__pycache__/duplicate_detector.cpython-314.pyc
ADDED
|
Binary file (2.82 kB). View file
|
|
|
services/ai_classifier.py
ADDED
|
@@ -0,0 +1,185 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import requests
|
| 2 |
+
import json
|
| 3 |
+
import os
|
| 4 |
+
import re
|
| 5 |
+
import logging
|
| 6 |
+
from domain.entities.article import Article
|
| 7 |
+
|
| 8 |
+
from services.lexicons import (
|
| 9 |
+
POSITIVE_WORDS,
|
| 10 |
+
NEGATIVE_WORDS,
|
| 11 |
+
CATEGORIES_MAP,
|
| 12 |
+
LOCATIONS_MAP,
|
| 13 |
+
MIDDLE_EAST_COUNTRIES,
|
| 14 |
+
STOP_WORDS
|
| 15 |
+
)
|
| 16 |
+
from services.local_nlp_classifier import LocalNLPClassifier
|
| 17 |
+
|
| 18 |
+
logger = logging.getLogger(__name__)
|
| 19 |
+
|
| 20 |
+
class HuggingFaceClassifier:
|
| 21 |
+
def __init__(self, api_key: str = None):
|
| 22 |
+
self.api_key = api_key or os.environ.get("HF_API_KEY")
|
| 23 |
+
# List of models to try in sequence if one hits rate limits or credit depletion
|
| 24 |
+
self.fallback_models = [
|
| 25 |
+
"Qwen/Qwen2.5-7B-Instruct",
|
| 26 |
+
"Qwen/Qwen2.5-Coder-7B-Instruct",
|
| 27 |
+
"deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B",
|
| 28 |
+
"meta-llama/Llama-3.1-8B-Instruct"
|
| 29 |
+
]
|
| 30 |
+
self.api_url = "https://router.huggingface.co/v1/chat/completions"
|
| 31 |
+
# Tier 2: Local NLP transformer models (loaded lazily on first use)
|
| 32 |
+
self._local_nlp = LocalNLPClassifier()
|
| 33 |
+
|
| 34 |
+
def enrich_article(self, article: Article) -> Article:
|
| 35 |
+
if not self.api_key:
|
| 36 |
+
logger.warning("No HF_API_KEY provided. Using local heuristic enrichment.")
|
| 37 |
+
return self._local_heuristic_fallback(article)
|
| 38 |
+
|
| 39 |
+
headers = {
|
| 40 |
+
"Authorization": f"Bearer {self.api_key}",
|
| 41 |
+
"Content-Type": "application/json"
|
| 42 |
+
}
|
| 43 |
+
|
| 44 |
+
prompt = """You are an article metadata enrichment service. Analyze the provided article title and summary and return ONLY valid JSON. Do not include any explanations or conversational text. Infer a single high-level category (Politics, Business, Technology, AI, Science, Health, Sports, Entertainment, Gaming, Culture, Education, Economy, Crime, Military, Weather, Environment, Travel, Lifestyle, Religion, Opinion, Local News, World), an optional subcategory, the primary region, country, and city if identifiable, 5–15 normalized tags, important keywords, named entities grouped into people, organizations, and locations, a sentiment (Positive, Negative, Neutral, Mixed), an importance score from 1 to 10 based on global significance, and an estimated readingTime in minutes. Do not invent facts that are not reasonably inferable from the title and summary. If information cannot be determined confidently, return null or an empty array. Return ONLY JSON matching this exact schema:
|
| 45 |
+
{
|
| 46 |
+
"category": "String",
|
| 47 |
+
"subcategory": "String",
|
| 48 |
+
"region": "String",
|
| 49 |
+
"country": "String",
|
| 50 |
+
"city": "String",
|
| 51 |
+
"tags": [],
|
| 52 |
+
"keywords": [],
|
| 53 |
+
"people": [],
|
| 54 |
+
"organizations": [],
|
| 55 |
+
"locations": [],
|
| 56 |
+
"sentiment": "String",
|
| 57 |
+
"importance": 0,
|
| 58 |
+
"readingTime": 0
|
| 59 |
+
}"""
|
| 60 |
+
|
| 61 |
+
# Try models in order until one succeeds
|
| 62 |
+
for model in self.fallback_models:
|
| 63 |
+
payload = {
|
| 64 |
+
"model": model,
|
| 65 |
+
"messages": [
|
| 66 |
+
{"role": "system", "content": prompt},
|
| 67 |
+
{"role": "user", "content": f"Title: {article.title}\nSummary: {article.summary}"}
|
| 68 |
+
],
|
| 69 |
+
"max_tokens": 1000,
|
| 70 |
+
"temperature": 0.1
|
| 71 |
+
}
|
| 72 |
+
|
| 73 |
+
try:
|
| 74 |
+
response = requests.post(self.api_url, headers=headers, json=payload, timeout=20)
|
| 75 |
+
if response.status_code == 200:
|
| 76 |
+
result = response.json()
|
| 77 |
+
content = result['choices'][0]['message']['content'].strip()
|
| 78 |
+
|
| 79 |
+
if content.startswith("```json"):
|
| 80 |
+
content = content[7:-3].strip()
|
| 81 |
+
elif content.startswith("```"):
|
| 82 |
+
content = content[3:-3].strip()
|
| 83 |
+
|
| 84 |
+
data = json.loads(content)
|
| 85 |
+
|
| 86 |
+
article.category = data.get('category') or article.category or "General"
|
| 87 |
+
article.subcategory = data.get('subcategory')
|
| 88 |
+
article.region = data.get('region')
|
| 89 |
+
article.country = data.get('country')
|
| 90 |
+
article.city = data.get('city')
|
| 91 |
+
article.tags = data.get('tags') or []
|
| 92 |
+
article.keywords = data.get('keywords') or []
|
| 93 |
+
article.people = data.get('people') or []
|
| 94 |
+
article.organizations = data.get('organizations') or []
|
| 95 |
+
article.locations = data.get('locations') or []
|
| 96 |
+
article.sentiment = data.get('sentiment') or "Neutral"
|
| 97 |
+
article.importance = data.get('importance') or 5
|
| 98 |
+
article.readingTime = data.get('readingTime') or self._estimate_reading_time(article)
|
| 99 |
+
|
| 100 |
+
return article
|
| 101 |
+
else:
|
| 102 |
+
logger.warning(f"HF Model {model} failed with status {response.status_code}. Trying next fallback...")
|
| 103 |
+
except Exception as e:
|
| 104 |
+
logger.warning(f"Exception trying model {model}: {e}. Trying next fallback...")
|
| 105 |
+
|
| 106 |
+
# If all AI models failed, try Tier 2: local NLP transformer models
|
| 107 |
+
logger.warning(f"All HF API models failed for article '{article.title[:50]}'. Trying Tier 2 local NLP models...")
|
| 108 |
+
if self._local_nlp.is_available():
|
| 109 |
+
try:
|
| 110 |
+
enriched = self._local_nlp.enrich_article(article)
|
| 111 |
+
logger.info("Tier 2 local NLP enrichment succeeded.")
|
| 112 |
+
return enriched
|
| 113 |
+
except Exception as e:
|
| 114 |
+
logger.warning(f"Tier 2 local NLP failed: {e}")
|
| 115 |
+
|
| 116 |
+
# Tier 3: If all else failed, use lexicon-based heuristic fallback so fields are never null
|
| 117 |
+
logger.error(f"All enrichment tiers failed for article '{article.title[:50]}'. Falling back to Tier 3 heuristics.")
|
| 118 |
+
return self._local_heuristic_fallback(article)
|
| 119 |
+
|
| 120 |
+
def _estimate_reading_time(self, article: Article) -> int:
|
| 121 |
+
text = f"{article.title} {article.summary or ''}"
|
| 122 |
+
# Strip simple HTML if present
|
| 123 |
+
text = re.sub(r'<[^>]+>', '', text)
|
| 124 |
+
words = len(text.split())
|
| 125 |
+
return max(1, round(words / 180))
|
| 126 |
+
|
| 127 |
+
def _local_heuristic_fallback(self, article: Article) -> Article:
|
| 128 |
+
"""High-precision rule-based NLP fallback using modular lexicons."""
|
| 129 |
+
full_text = f"{article.title} {article.summary or ''}".lower()
|
| 130 |
+
|
| 131 |
+
# 1. Category & Subcategory Detection
|
| 132 |
+
if not article.category or article.category in ["General", "Middle East", "Egypt", "World", "North Africa"]:
|
| 133 |
+
for cat_name, (keywords, subcat) in CATEGORIES_MAP.items():
|
| 134 |
+
if any(kw in full_text for kw in keywords):
|
| 135 |
+
article.category = cat_name
|
| 136 |
+
article.subcategory = subcat
|
| 137 |
+
break
|
| 138 |
+
if not article.category:
|
| 139 |
+
article.category = "General"
|
| 140 |
+
|
| 141 |
+
# 2. Entity & Location Extraction
|
| 142 |
+
detected_locations = []
|
| 143 |
+
for loc_kw, loc_name in LOCATIONS_MAP.items():
|
| 144 |
+
if loc_kw in full_text:
|
| 145 |
+
if loc_name not in detected_locations:
|
| 146 |
+
detected_locations.append(loc_name)
|
| 147 |
+
|
| 148 |
+
article.locations = detected_locations if detected_locations else ["Global"]
|
| 149 |
+
if detected_locations:
|
| 150 |
+
article.country = detected_locations[0]
|
| 151 |
+
article.region = "Middle East" if any(l in MIDDLE_EAST_COUNTRIES for l in detected_locations) else "Global"
|
| 152 |
+
|
| 153 |
+
# 3. Clean Keyword & Tag Extraction
|
| 154 |
+
clean_words = [
|
| 155 |
+
re.sub(r'[^\w\s]', '', w) for w in article.title.split()
|
| 156 |
+
if len(w) > 3 and w.lower() not in STOP_WORDS
|
| 157 |
+
]
|
| 158 |
+
|
| 159 |
+
article.keywords = list(dict.fromkeys(clean_words[:6]))
|
| 160 |
+
article.tags = list(dict.fromkeys(clean_words[:10]))
|
| 161 |
+
|
| 162 |
+
# 4. Lexicon Sentiment Analysis
|
| 163 |
+
pos_count = sum(1 for w in POSITIVE_WORDS if w in full_text)
|
| 164 |
+
neg_count = sum(1 for w in NEGATIVE_WORDS if w in full_text)
|
| 165 |
+
|
| 166 |
+
if pos_count > neg_count:
|
| 167 |
+
article.sentiment = "Positive"
|
| 168 |
+
elif neg_count > pos_count:
|
| 169 |
+
article.sentiment = "Negative"
|
| 170 |
+
else:
|
| 171 |
+
article.sentiment = "Neutral"
|
| 172 |
+
|
| 173 |
+
# 5. Dynamic Importance Scoring (1 - 10)
|
| 174 |
+
high_impact_terms = ["عاجل", "قمة", "حرب", "وفاة", "رئيس", "كوارث", "breaking", "war", "president", "disaster", "crisis", "urgent"]
|
| 175 |
+
importance_score = 5
|
| 176 |
+
if any(term in full_text for term in high_impact_terms):
|
| 177 |
+
importance_score += 3
|
| 178 |
+
if len(detected_locations) > 1:
|
| 179 |
+
importance_score += 1
|
| 180 |
+
article.importance = min(10, importance_score)
|
| 181 |
+
|
| 182 |
+
# 6. Reading Time Estimation
|
| 183 |
+
article.readingTime = self._estimate_reading_time(article)
|
| 184 |
+
|
| 185 |
+
return article
|
services/duplicate_detector.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import logging
|
| 2 |
+
from typing import List
|
| 3 |
+
from domain.entities.article import Article
|
| 4 |
+
from domain.repositories.article_repository import ArticleRepository
|
| 5 |
+
|
| 6 |
+
logger = logging.getLogger(__name__)
|
| 7 |
+
|
| 8 |
+
class DuplicateDetector:
|
| 9 |
+
def __init__(self, repository: ArticleRepository):
|
| 10 |
+
self.repository = repository
|
| 11 |
+
|
| 12 |
+
def filter_new_articles(self, articles: List[Article]) -> List[Article]:
|
| 13 |
+
"""Robust multi-level duplicate detection (URL Hash, In-Batch Seen, Title Hash)."""
|
| 14 |
+
new_articles = []
|
| 15 |
+
seen_ids = set()
|
| 16 |
+
seen_title_hashes = set()
|
| 17 |
+
skipped_count = 0
|
| 18 |
+
|
| 19 |
+
for article in articles:
|
| 20 |
+
# 1. Normalize ID and Title Hash
|
| 21 |
+
article_id = article.id
|
| 22 |
+
title_hash = Article.generate_title_hash(article.title)
|
| 23 |
+
|
| 24 |
+
# 2. In-batch Duplicate Check (prevents duplicates inside the same scrape run)
|
| 25 |
+
if article_id in seen_ids or title_hash in seen_title_hashes:
|
| 26 |
+
logger.debug(f"In-batch duplicate detected: {article.title[:40]}")
|
| 27 |
+
skipped_count += 1
|
| 28 |
+
continue
|
| 29 |
+
|
| 30 |
+
# 3. Database Check (checks if article ID already exists in Firestore)
|
| 31 |
+
if self.repository.exists(article_id):
|
| 32 |
+
logger.debug(f"Article exists in Firestore: {article.title[:40]}")
|
| 33 |
+
skipped_count += 1
|
| 34 |
+
# Mark as seen so we don't check Firestore again if it reappears in feed
|
| 35 |
+
seen_ids.add(article_id)
|
| 36 |
+
seen_title_hashes.add(title_hash)
|
| 37 |
+
continue
|
| 38 |
+
|
| 39 |
+
# 4. Mark as unique and new
|
| 40 |
+
seen_ids.add(article_id)
|
| 41 |
+
seen_title_hashes.add(title_hash)
|
| 42 |
+
new_articles.append(article)
|
| 43 |
+
|
| 44 |
+
logger.info(f"Duplicate Detection complete: {len(new_articles)} new, {skipped_count} duplicates skipped.")
|
| 45 |
+
return new_articles
|
services/lexicons/__init__.py
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from .sentiment_lexicon import POSITIVE_WORDS, NEGATIVE_WORDS
|
| 2 |
+
from .categories_lexicon import CATEGORIES_MAP
|
| 3 |
+
from .locations_lexicon import LOCATIONS_MAP, MIDDLE_EAST_COUNTRIES
|
| 4 |
+
from .stopwords_lexicon import STOP_WORDS
|
| 5 |
+
from .images_lexicon import CATEGORY_IMAGES_MAP
|
| 6 |
+
|
| 7 |
+
__all__ = [
|
| 8 |
+
"POSITIVE_WORDS",
|
| 9 |
+
"NEGATIVE_WORDS",
|
| 10 |
+
"CATEGORIES_MAP",
|
| 11 |
+
"LOCATIONS_MAP",
|
| 12 |
+
"MIDDLE_EAST_COUNTRIES",
|
| 13 |
+
"STOP_WORDS",
|
| 14 |
+
"CATEGORY_IMAGES_MAP"
|
| 15 |
+
]
|
services/lexicons/__pycache__/__init__.cpython-314.pyc
ADDED
|
Binary file (584 Bytes). View file
|
|
|
services/lexicons/__pycache__/categories_lexicon.cpython-314.pyc
ADDED
|
Binary file (13.9 kB). View file
|
|
|
services/lexicons/__pycache__/images_lexicon.cpython-314.pyc
ADDED
|
Binary file (2.25 kB). View file
|
|
|
services/lexicons/__pycache__/locations_lexicon.cpython-314.pyc
ADDED
|
Binary file (7.37 kB). View file
|
|
|
services/lexicons/__pycache__/sentiment_lexicon.cpython-314.pyc
ADDED
|
Binary file (8.25 kB). View file
|
|
|
services/lexicons/__pycache__/stopwords_lexicon.cpython-314.pyc
ADDED
|
Binary file (3.32 kB). View file
|
|
|
services/lexicons/categories_lexicon.py
ADDED
|
@@ -0,0 +1,135 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Extensive Category and Subcategory Keyword Mappings for Arabic & English News NLP
|
| 2 |
+
|
| 3 |
+
CATEGORIES_MAP = {
|
| 4 |
+
"Sports": (
|
| 5 |
+
[
|
| 6 |
+
"كرة", "مباراة", "دوري", "بطولة", "الأهلي", "الزمالك", "صلاح", "رياضة", "منتخب", "هداف",
|
| 7 |
+
"ملعب", "مدرب", "صفقة", "انتقالات", "كأس", "كرة قدم", "سلة", "كرة يد", "تنس", "أولمبياد",
|
| 8 |
+
"ريال مدريد", "برشلونة", "مانشستر", "ليفربول", "دوري أبطال", "كأس العالم", "رونالدو", "ميسي",
|
| 9 |
+
"football", "soccer", "match", "league", "cup", "nba", "tennis", "stadium", "coach",
|
| 10 |
+
"tournament", "player", "fifa", "uefa", "champion", "olympics", "nfl", "formula1", "f1",
|
| 11 |
+
"real madrid", "barcelona", "liverpool", "champions league", "world cup", "messi", "ronaldo"
|
| 12 |
+
],
|
| 13 |
+
"Football"
|
| 14 |
+
),
|
| 15 |
+
"Technology": (
|
| 16 |
+
[
|
| 17 |
+
"ذكاء اصطناعي", "تقنية", "هاتف", "تطبيق", "أبل", "سامسونج", "جوجل", "برمجيات", "روبوت",
|
| 18 |
+
"سحابة", "أمن سيبراني", "معالج", "آيفون", "أندرويد", "تحديث", "شاشة", "حاسوب", "ابتكار",
|
| 19 |
+
"شبكات", "5g", "رقائق", "سيليكون", "أنظمة", "خوارزمية", "شات جي بي تي", "إنترنت",
|
| 20 |
+
"ai", "artificial intelligence", "tech", "technology", "apple", "google", "software",
|
| 21 |
+
"cyber", "cybersecurity", "iphone", "android", "cloud", "robotics", "processor", "gadget",
|
| 22 |
+
"app", "application", "startup", "microsoft", "semiconductor", "chip", "chatgpt", "openai"
|
| 23 |
+
],
|
| 24 |
+
"Tech News"
|
| 25 |
+
),
|
| 26 |
+
"Gaming": (
|
| 27 |
+
[
|
| 28 |
+
"لعبة", "العاب", "ألعاب", "بلايستيشن", "إكس بوكس", "قيمينق", "جيمرز", "نينتندو", "ستيم",
|
| 29 |
+
"إي سبورتس", "مطور ألعاب", "تسريبات لعبة", "تحديث لعبة", "سوني", "مهمة", "بطولة العاب",
|
| 30 |
+
"ببجي", "فورتنايت", "كول اوف ديوتي", "فيفا 24", "العاب فيديو",
|
| 31 |
+
"game", "gaming", "playstation", "xbox", "nintendo", "esports", "steam", "gamer",
|
| 32 |
+
"ps5", "unreal engine", "graphics", "gameplay", "gta", "fortnite", "call of duty", "rpg", "roblox"
|
| 33 |
+
],
|
| 34 |
+
"Gaming"
|
| 35 |
+
),
|
| 36 |
+
"Business": (
|
| 37 |
+
[
|
| 38 |
+
"اقتصاد", "أسهم", "بورصة", "دولار", "تضخم", "شركات", "أرباح", "استثمار", "بنك", "مركزي",
|
| 39 |
+
"تجارة", "أسواق", "نفط", "غاز", "عملات", "عقارات", "استثمارات", "ميزانية", "صندوق النقد",
|
| 40 |
+
"فائدة", "بنك دولي", "استحواذ", "اندماج", "تداول", "ذهب", "سندات", "مشاريع",
|
| 41 |
+
"economy", "market", "markets", "stock", "stocks", "dollar", "inflation", "bank", "banking",
|
| 42 |
+
"finance", "financial", "investment", "investor", "revenue", "trade", "crypto", "bitcoin",
|
| 43 |
+
"real estate", "fed", "central bank", "gdp", "gold", "acquisition", "bonds"
|
| 44 |
+
],
|
| 45 |
+
"Finance"
|
| 46 |
+
),
|
| 47 |
+
"Health": (
|
| 48 |
+
[
|
| 49 |
+
"صحة", "مرض", "علاج", "طبيب", "أطباء", "مستشفى", "مستشفيات", "فيروس", "لقاح", "دواء",
|
| 50 |
+
"أدوية", "سرطان", "مناعة", "قلب", "حمية", "تغذية", "وباء", "منظمة الصحة", "عيادة",
|
| 51 |
+
"جراحة", "صحة نفسية", "أعراض", "تشخيص", "وقاية", "سمنة",
|
| 52 |
+
"health", "virus", "medical", "doctor", "hospital", "medicine", "vaccine", "cancer",
|
| 53 |
+
"disease", "epidemic", "pandemic", "wellness", "diet", "nutrition", "pharma", "who",
|
| 54 |
+
"surgery", "mental health", "symptoms", "diagnosis", "prevention"
|
| 55 |
+
],
|
| 56 |
+
"Medical"
|
| 57 |
+
),
|
| 58 |
+
"Science": (
|
| 59 |
+
[
|
| 60 |
+
"فضاء", "كوكب", "علماء", "مريخ", "ناسا", "بيئة", "مناخ", "أبحاث", "مجرة", "تلسكوب",
|
| 61 |
+
"احتباس حراري", "فيزياء", "كيمياء", "طبيعة", "احافير", "ذرة", "طاقة تجددة", "اكتشاف",
|
| 62 |
+
"علمي", "تجربة", "مختبر", "أحياء", "جينات", "وراثة",
|
| 63 |
+
"science", "space", "nasa", "planet", "climate", "research", "galaxy", "telescope",
|
| 64 |
+
"astronomy", "physics", "biology", "fossil", "atom", "renewable energy", "nature",
|
| 65 |
+
"discovery", "scientific", "laboratory", "genetics", "dna"
|
| 66 |
+
],
|
| 67 |
+
"Space & Research"
|
| 68 |
+
),
|
| 69 |
+
"Food": (
|
| 70 |
+
[
|
| 71 |
+
"وجبة", "طعام", "وصفة", "وصفات", "مطعم", "مطاعم", "طهي", "أكل", "مطبخ", "شيف", "عشاء",
|
| 72 |
+
"غداء", "إفطار", "مكونات", "طبق", "حلويات", "مخبوزات", "طبخ", "حلويات شرقية", "مشروبات",
|
| 73 |
+
"food", "recipe", "recipes", "cooking", "kitchen", "dish", "dishes", "restaurant",
|
| 74 |
+
"chef", "meal", "cuisine", "flavor", "bakery", "dessert", "dining", "flavor", "baking"
|
| 75 |
+
],
|
| 76 |
+
"Culinary"
|
| 77 |
+
),
|
| 78 |
+
"Entertainment": (
|
| 79 |
+
[
|
| 80 |
+
"فيلم", "أفلام", "مسلسل", "مسلسلات", "سينما", "موسيقى", "فنان", "فنانة", "مهرجان", "أغنية",
|
| 81 |
+
"هوليوود", "دراما", "كوميديا", "شباك التذاكر", "أوسكار", "عرض أول", "حفلة", "أغاني",
|
| 82 |
+
"تريند", "مشاهير", "نجوم", "بوستر", "تريلر", "برومو",
|
| 83 |
+
"movie", "movies", "film", "cinema", "music", "actor", "actress", "hollywood", "series",
|
| 84 |
+
"tv show", "box office", "oscars", "grammy", "concert", "celebrity", "drama", "trailer", "trend"
|
| 85 |
+
],
|
| 86 |
+
"Movies & TV"
|
| 87 |
+
),
|
| 88 |
+
"Politics": (
|
| 89 |
+
[
|
| 90 |
+
"رئيس", "وزير", "وزراء", "حكومة", "برلمان", "انتخابات", "سفارة", "سفير", "دبلوماسية",
|
| 91 |
+
"مفاوضات", "معاهدة", "مجلس الأمن", "الأمم المتحدة", "حزب", "سياسة", "خارجية", "دفاع",
|
| 92 |
+
"عسكري", "جيش", "قوات", "سياسي", "قمة", "مؤتمر صحفي", "سياسيون",
|
| 93 |
+
"president", "minister", "government", "parliament", "election", "elections", "diplomacy",
|
| 94 |
+
"diplomatic", "ambassador", "treaty", "un security council", "senate", "congress", "policy",
|
| 95 |
+
"military", "army", "forces", "summit", "geopolitics"
|
| 96 |
+
],
|
| 97 |
+
"Government"
|
| 98 |
+
),
|
| 99 |
+
"Education": (
|
| 100 |
+
[
|
| 101 |
+
"تعليم", "مدرسة", "مدارس", "جامعة", "جامعات", "طلاب", "طالب", "امتحانات", "امتحان",
|
| 102 |
+
"نتيجة", "تنسيق", "وزارة التربية والتعليم", "بكالوريا", "ثانوية عامة", "دراسة", "منحة",
|
| 103 |
+
"education", "school", "schools", "university", "student", "students", "exam", "exams",
|
| 104 |
+
"degree", "scholarship", "academy", "curriculum", "campus"
|
| 105 |
+
],
|
| 106 |
+
"Education"
|
| 107 |
+
),
|
| 108 |
+
"Crime": (
|
| 109 |
+
[
|
| 110 |
+
"جريمة", "جرائم", "شرطة", "قبض", "اعتقال", "مخدرات", "تحقيق", "نيابة", "محكمة", "قضاء",
|
| 111 |
+
"احتيال", "سرقة", "سطو", "قتل", "عصابة", "تهريب", "قاضي",
|
| 112 |
+
"crime", "police", "arrest", "arrested", "drugs", "investigation", "court", "judge",
|
| 113 |
+
"fraud", "theft", "robbery", "murder", "gang", "smuggling", "trial"
|
| 114 |
+
],
|
| 115 |
+
"Crime & Justice"
|
| 116 |
+
),
|
| 117 |
+
"Weather": (
|
| 118 |
+
[
|
| 119 |
+
"طقس", "مناخ", "حرارة", "أمطار", "مطر", "عاصفة", "عواصف", "ثلوج", "رياح", "أرصاد",
|
| 120 |
+
"موجة حارة", "إعصار", "سيول", "درجات الحرارة",
|
| 121 |
+
"weather", "forecast", "temperature", "rain", "storm", "snow", "wind", "hurricane",
|
| 122 |
+
"heatwave", "blizzard", "meteorology"
|
| 123 |
+
],
|
| 124 |
+
"Weather & Climate"
|
| 125 |
+
),
|
| 126 |
+
"Travel": (
|
| 127 |
+
[
|
| 128 |
+
"سفر", "سياحة", "فندق", "فنادق", "طيران", "رحلة", "رحلات", "وجهة", "تأشيرة", "فيزا",
|
| 129 |
+
"مطار", "معالم", "شاطئ", "منتجع",
|
| 130 |
+
"travel", "tourism", "hotel", "hotels", "flight", "flights", "destination", "visa",
|
| 131 |
+
"airport", "beach", "resort", "vacation"
|
| 132 |
+
],
|
| 133 |
+
"Travel & Leisure"
|
| 134 |
+
)
|
| 135 |
+
}
|
services/lexicons/images_lexicon.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# High-Resolution Category Fallback Cover Images (Royalty-Free Unsplash Assets)
|
| 2 |
+
|
| 3 |
+
CATEGORY_IMAGES_MAP = {
|
| 4 |
+
"Technology": "https://images.unsplash.com/photo-1518770660439-4636190af475?auto=format&fit=crop&w=1000&q=80",
|
| 5 |
+
"Gaming": "https://images.unsplash.com/photo-1538481199705-c710c4e965fc?auto=format&fit=crop&w=1000&q=80",
|
| 6 |
+
"Sports": "https://images.unsplash.com/photo-1461896836934-ffe607ba8211?auto=format&fit=crop&w=1000&q=80",
|
| 7 |
+
"Business": "https://images.unsplash.com/photo-1611974789855-9c2a0a7236a3?auto=format&fit=crop&w=1000&q=80",
|
| 8 |
+
"Health": "https://images.unsplash.com/photo-1505751172876-fa1923c5c528?auto=format&fit=crop&w=1000&q=80",
|
| 9 |
+
"Science": "https://images.unsplash.com/photo-1451187580459-43490279c0fa?auto=format&fit=crop&w=1000&q=80",
|
| 10 |
+
"Food": "https://images.unsplash.com/photo-1504674900247-0877df9cc836?auto=format&fit=crop&w=1000&q=80",
|
| 11 |
+
"Entertainment": "https://images.unsplash.com/photo-1489599849927-2ee91cede3ba?auto=format&fit=crop&w=1000&q=80",
|
| 12 |
+
"Politics": "https://images.unsplash.com/photo-1541872703-74c5e44368f9?auto=format&fit=crop&w=1000&q=80",
|
| 13 |
+
"Education": "https://images.unsplash.com/photo-1523240795612-9a054b0db644?auto=format&fit=crop&w=1000&q=80",
|
| 14 |
+
"Crime": "https://images.unsplash.com/photo-1589829545856-d10d557cf95f?auto=format&fit=crop&w=1000&q=80",
|
| 15 |
+
"Weather": "https://images.unsplash.com/photo-1516912481808-3406841bd33c?auto=format&fit=crop&w=1000&q=80",
|
| 16 |
+
"Travel": "https://images.unsplash.com/photo-1488646953014-85cb44e25828?auto=format&fit=crop&w=1000&q=80",
|
| 17 |
+
"Middle East": "https://images.unsplash.com/photo-1512453979798-5ea266f8880c?auto=format&fit=crop&w=1000&q=80",
|
| 18 |
+
"Egypt": "https://images.unsplash.com/photo-1572252821128-d53b16d39d5e?auto=format&fit=crop&w=1000&q=80",
|
| 19 |
+
"World": "https://images.unsplash.com/photo-1526778548025-fa2f459cd5c1?auto=format&fit=crop&w=1000&q=80",
|
| 20 |
+
"General": "https://images.unsplash.com/photo-1504711434969-e33886168f5c?auto=format&fit=crop&w=1000&q=80"
|
| 21 |
+
}
|
services/lexicons/locations_lexicon.py
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Comprehensive Global and Middle East Geographic Gazetteer Mapping
|
| 2 |
+
|
| 3 |
+
LOCATIONS_MAP = {
|
| 4 |
+
# Egypt (Governorates & Cities)
|
| 5 |
+
"مصر": "Egypt", "القاهرة": "Cairo", "الإسكندرية": "Alexandria", "الجيزة": "Giza",
|
| 6 |
+
"سيناء": "Sinai", "شرم الشيخ": "Sharm El Sheikh", "العلمين": "El Alamein", "أسوان": "Aswan",
|
| 7 |
+
"الأقصر": "Luxor", "مطروح": "Matrouh", "بورسعيد": "Port Said", "السويس": "Suez",
|
| 8 |
+
"طنطا": "Tanta", "المنصورة": "Mansoura", "الزقازيق": "Zagazig", "الإسماعيلية": "Ismailia",
|
| 9 |
+
"الفيوم": "Faiyum", "بني سويف": "Beni Suef", "المنيا": "Minya", "أسيوط": "Asyut",
|
| 10 |
+
"سوهاج": "Sohag", "قنا": "Qena", "الغردقة": "Hurghada", "دمياط": "Damietta",
|
| 11 |
+
"egypt": "Egypt", "cairo": "Cairo", "alexandria": "Alexandria", "giza": "Giza",
|
| 12 |
+
|
| 13 |
+
# Palestine & Levant
|
| 14 |
+
"غزة": "Gaza", "فلسطين": "Palestine", "القدس": "Jerusalem", "الضفة": "West Bank",
|
| 15 |
+
"رام الله": "Ramallah", "رفح": "Rafah", "خان يونس": "Khan Younis", "جنين": "Jenin",
|
| 16 |
+
"نابلس": "Nablus", "بيروت": "Beirut", "لبنان": "Lebanon", "دمشق": "Damascus", "حلب": "Aleppo",
|
| 17 |
+
"سوريا": "Syria", "عمان": "Amman", "الأردن": "Jordan", "إربد": "Irbid", "الزرعاء": "Zarqaa",
|
| 18 |
+
"gaza": "Gaza", "palestine": "Palestine", "jerusalem": "Jerusalem", "lebanon": "Lebanon",
|
| 19 |
+
"syria": "Syria", "jordan": "Jordan", "beirut": "Beirut", "damascus": "Damascus",
|
| 20 |
+
|
| 21 |
+
# Gulf & Arabian Peninsula
|
| 22 |
+
"السعودية": "Saudi Arabia", "الرياض": "Riyadh", "جدة": "Jeddah", "مكة": "Mekkah", "المدينة": "Madinah",
|
| 23 |
+
"الدمام": "Dammam", "الخبر": "Khobar", "الإمارات": "UAE", "دبي": "Dubai", "أبوظبي": "Abu Dhabi",
|
| 24 |
+
"الشارقة": "Sharjah", "قطر": "Qatar", "الدوحة": "Doha", "الكويت": "Kuwait", "البحرين": "Bahrain",
|
| 25 |
+
"المنامة": "Manama", "عمان": "Oman", "مسقط": "Muscat", "اليمن": "Yemen", "صنعاء": "Sanaa", "عدن": "Aden",
|
| 26 |
+
"saudi arabia": "Saudi Arabia", "riyadh": "Riyadh", "dubai": "Dubai", "uae": "UAE",
|
| 27 |
+
"qatar": "Qatar", "doha": "Doha", "kuwait": "Kuwait", "abu dhabi": "Abu Dhabi",
|
| 28 |
+
|
| 29 |
+
# North Africa
|
| 30 |
+
"المغرب": "Morocco", "الرباط": "Rabat", "الدار البيضاء": "Casablanca", "مراكش": "Marrakech",
|
| 31 |
+
"فاس": "Fes", "طنجة": "Tangier", "الجزائر": "Algeria", "وهران": "Oran", "قسنطينة": "Constantine",
|
| 32 |
+
"تونس": "Tunisia", "صفاقس": "Sfax", "ليبيا": "Libya", "طرابلس": "Tripoli", "بنغازي": "Benghazi",
|
| 33 |
+
"السودان": "Sudan", "الخرطوم": "Khartoum", "أم درمان": "Omdurman",
|
| 34 |
+
"morocco": "Morocco", "algeria": "Algeria", "tunisia": "Tunisia", "libya": "Libya", "sudan": "Sudan",
|
| 35 |
+
|
| 36 |
+
# Iraq
|
| 37 |
+
"العراق": "Iraq", "بغداد": "Baghdad", "أربيل": "Erbil", "البصرة": "Basra", "الموصل": "Mosul",
|
| 38 |
+
"iraq": "Iraq", "baghdad": "Baghdad",
|
| 39 |
+
|
| 40 |
+
# Global Powers & Major Capitals
|
| 41 |
+
"واشنطن": "Washington", "أمريكا": "USA", "الولايات المتحدة": "USA", "نيويورك": "New York",
|
| 42 |
+
"شيكاغو": "Chicago", "كاليفورنيا": "California", "تكساس": "Texas", "فلوريدا": "Florida",
|
| 43 |
+
"باريس": "Paris", "فرنسا": "France", "مارسيليا": "Marseille", "لندن": "London", "بريطانيا": "UK",
|
| 44 |
+
"إنجلترا": "UK", "مانشستر": "Manchester", "برلين": "Berlin", "ألمانيا": "Germany", "ميونيخ": "Munich",
|
| 45 |
+
"بكين": "Beijing", "الصين": "China", "شنغهاي": "Shanghai", "موسكو": "Moscow", "روسيا": "Russia",
|
| 46 |
+
"طوكيو": "Tokyo", "اليابان": "Japan", "روما": "Rome", "إيطاليا": "Italy", "ميلانو": "Milan",
|
| 47 |
+
"مدريد": "Madrid", "إسبانيا": "Spain", "برشلونة": "Barcelona", "أنقرة": "Ankara", "تركيا": "Turkey",
|
| 48 |
+
"إسطنبول": "Istanbul", "كييف": "Kyiv", "أوكرانيا": "Ukraine", "سول": "Seoul", "كوريا": "Korea",
|
| 49 |
+
"نيودلهي": "New Delhi", "الهند": "India", "برازيليا": "Brasilia", "البرازيل": "Brazil",
|
| 50 |
+
"سدني": "Sydney", "أستراليا": "Australia", "تورونتو": "Toronto", "كندا": "Canada",
|
| 51 |
+
"usa": "USA", "us": "USA", "uk": "UK", "china": "China", "russia": "Russia", "france": "France",
|
| 52 |
+
"germany": "Germany", "japan": "Japan", "turkey": "Turkey", "ukraine": "Ukraine", "canada": "Canada"
|
| 53 |
+
}
|
| 54 |
+
|
| 55 |
+
MIDDLE_EAST_COUNTRIES = {
|
| 56 |
+
"Egypt", "Gaza", "Palestine", "Saudi Arabia", "UAE", "Qatar", "Kuwait", "Lebanon",
|
| 57 |
+
"Syria", "Jordan", "Iraq", "Yemen", "Oman", "Bahrain", "Sudan", "Libya", "Algeria",
|
| 58 |
+
"Morocco", "Tunisia"
|
| 59 |
+
}
|
services/lexicons/sentiment_lexicon.py
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Extensive sentiment lexicons for Arabic and English NLP classification
|
| 2 |
+
|
| 3 |
+
POSITIVE_WORDS = {
|
| 4 |
+
# Arabic Positive Cues
|
| 5 |
+
"فوز", "نجاح", "انتصار", "تألق", "تفوق", "نمو", "ازدهار", "أرباح", "ربح", "استقرار",
|
| 6 |
+
"اتفاق", "إنجاز", "افتتاح", "مكاسب", "تبرع", "تبرعات", "مساعدة", "حليف", "صعود", "انتعاش",
|
| 7 |
+
"ابتكار", "تطوير", "حسم", "ريادة", "امتياز", "سلام", "تعافي", "تسهيل", "دعم", "شراكة",
|
| 8 |
+
"تأهل", "ذهبية", "فضية", "برونزية", "ترقية", "جائزة", "تكريم", "مبادرة", "هدنة", "إيجابي",
|
| 9 |
+
"تقدم", "اعتماد", "موافقة", "تسهيلات", "شفاء", "علاج", "بشرى", "نجاحات", "تفاؤل", "فرحة",
|
| 10 |
+
"مكافأة", "تحسين", "توسيع", "إنقاذ", "صفقة", "عقد", "استثمار", "تسهيل", "إشادة", "ثراء",
|
| 11 |
+
"خير", "بركة", "بهجة", "سعادة", "انتصارات", "مكتسبات", "تكاتف", "تضامن", "مرونة", "تأهيل",
|
| 12 |
+
"إعفاء", "تسوية", "تسامح", "إنعاش", "بشائر", "طفرة", "تنمية", "إعادة إعمار", "جاهزية",
|
| 13 |
+
|
| 14 |
+
# English Positive Cues
|
| 15 |
+
"win", "winning", "winner", "victory", "triumph", "success", "successful", "growth", "grow",
|
| 16 |
+
"profit", "profitable", "boom", "booming", "recovery", "recover", "peace", "agreement",
|
| 17 |
+
"breakthrough", "milestone", "boost", "boosting", "gain", "gains", "innovation", "innovative",
|
| 18 |
+
"positive", "flourish", "thrive", "thriving", "achievement", "award", "awarded", "hero",
|
| 19 |
+
"heroic", "stability", "stable", "champion", "champions", "championship", "progress",
|
| 20 |
+
"partnership", "soar", "soaring", "upgrade", "advance", "advancement", "relief", "cure",
|
| 21 |
+
"prosper", "prosperity", "benefit", "beneficial", "glory", "glorious", "praise", "praised",
|
| 22 |
+
"accord", "reconciliation", "revival", "revive", "opportunity", "record", "surge", "surging",
|
| 23 |
+
"optimism", "optimistic", "solidarity", "resilience", "fund", "funded", "grant", "granted"
|
| 24 |
+
}
|
| 25 |
+
|
| 26 |
+
NEGATIVE_WORDS = {
|
| 27 |
+
# Arabic Negative Cues
|
| 28 |
+
"مقتل", "قتيل", "وفاة", "وفيات", "أزمة", "أزمات", "تراجع", "انخفاض", "خسارة", "خسائر",
|
| 29 |
+
"انهيار", "جريمة", "جرائم", "حادث", "حوادث", "كارثة", "كوارث", "تحذير", "عقوبات", "إفلاس",
|
| 30 |
+
"تهديد", "خيانة", "سرقة", "قلق", "زلزال", "فيضان", "حريق", "حرائق", "قتلى", "جرحى",
|
| 31 |
+
"مصابين", "إصابة", "حرب", "حروب", "قصف", "هجوم", "انفجار", "انفجارات", "توتر", "مأساة",
|
| 32 |
+
"ضحايا", "نزاع", "اعتقال", "اشتباكات", "شهداء", "عنف", "دمار", "تدمير", "مغادرة", "احتجاجات",
|
| 33 |
+
"نزوح", "لاجئين", "فقر", "مجاعة", "غلاء", "ركود", "إغلاق", "إلغاء", "إضراب", "اتهام",
|
| 34 |
+
"ادانة", "إدانة", "خراب", "سقوط", "نزيف", "تدهور", "مخاطر", "خطر", "فشل", "اضطراب",
|
| 35 |
+
"فضيحة", "فساد", "اختلاس", "احتيال", "مداهمة", "اعتقالات", "حصار", "احتلال", "قمع",
|
| 36 |
+
|
| 37 |
+
# English Negative Cues
|
| 38 |
+
"death", "dead", "died", "kill", "killed", "killing", "murder", "attack", "attacks", "war",
|
| 39 |
+
"warfare", "crisis", "crash", "crashed", "decline", "declining", "loss", "losses", "collapse",
|
| 40 |
+
"crime", "criminal", "warning", "warn", "threat", "threatening", "disaster", "disastrous",
|
| 41 |
+
"accident", "explosion", "explosions", "sanction", "sanctions", "inflation", "tragedy",
|
| 42 |
+
"tragic", "casualty", "casualties", "scandal", "fatal", "fatality", "strike", "strikes",
|
| 43 |
+
"conflict", "violence", "violent", "injury", "injured", "ruin", "ruined", "protest", "riot",
|
| 44 |
+
"poverty", "famine", "recession", "cancel", "cancelled", "fraud", "corruption", "arrest",
|
| 45 |
+
"arrested", "siege", "occupation", "devastation", "devastating", "slump", "downfall", "fear"
|
| 46 |
+
}
|
services/lexicons/stopwords_lexicon.py
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Comprehensive Arabic & English Stopwords for NLP Tokenization & Tagging
|
| 2 |
+
|
| 3 |
+
STOP_WORDS = {
|
| 4 |
+
# Arabic Stopwords
|
| 5 |
+
"في", "من", "على", "عن", "أن", "إن", "إلى", "مع", "هذا", "هذه", "تم", "بعد", "قبل",
|
| 6 |
+
"كان", "كانت", "يكون", "تكون", "أو", "أم", "ثم", "حتى", "التي", "الذي", "الذين",
|
| 7 |
+
"اللذين", "اللتين", "بين", "حول", "ضد", "نحو", "كل", "جميع", "بعض", "غير", "ذات",
|
| 8 |
+
"ما", "ماذا", "منذ", "خلال", "عند", "أمام", "خلف", "تحت", "فوق", "كما", "لكن",
|
| 9 |
+
"غير", "أيضا", "فقط", "جدا", "قد", "لقد", "حيث", "إذا", "سوف", "سيكون", "اليوم",
|
| 10 |
+
|
| 11 |
+
# English Stopwords
|
| 12 |
+
"the", "a", "an", "in", "on", "at", "to", "for", "of", "with", "and", "is", "are",
|
| 13 |
+
"was", "were", "be", "been", "being", "have", "has", "had", "do", "does", "did",
|
| 14 |
+
"but", "or", "because", "as", "until", "while", "about", "against", "between",
|
| 15 |
+
"into", "through", "during", "before", "after", "above", "below", "from", "up",
|
| 16 |
+
"down", "in", "out", "over", "under", "again", "further", "then", "once", "here",
|
| 17 |
+
"there", "when", "where", "why", "how", "all", "any", "both", "each", "few", "more",
|
| 18 |
+
"most", "other", "some", "such", "no", "nor", "not", "only", "own", "same", "so",
|
| 19 |
+
"than", "too", "very", "s", "t", "can", "will", "just", "don", "should", "now"
|
| 20 |
+
}
|
services/local_nlp_classifier.py
ADDED
|
@@ -0,0 +1,399 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Tier 2 Local NLP Classifier — Pre-trained Transformer Models.
|
| 3 |
+
|
| 4 |
+
This module provides near-LLM-quality enrichment using locally-downloaded
|
| 5 |
+
HuggingFace transformer models. It sits BETWEEN the HuggingFace API (Tier 1)
|
| 6 |
+
and the lexicon-based heuristics (Tier 3) in the fallback chain.
|
| 7 |
+
|
| 8 |
+
Models used:
|
| 9 |
+
- Category: facebook/bart-large-mnli (zero-shot classification, multilingual)
|
| 10 |
+
- Sentiment: CAMeL-Lab/bert-base-arabic-camelbert-msa-sentiment (Arabic)
|
| 11 |
+
distilbert-base-uncased-finetuned-sst-2-english (English)
|
| 12 |
+
- NER: CAMeL-Lab/bert-base-arabic-camelbert-msa-ner (Arabic)
|
| 13 |
+
dslim/bert-base-NER (English)
|
| 14 |
+
- Keywords: KeyBERT with paraphrase-multilingual-MiniLM-L12-v2
|
| 15 |
+
"""
|
| 16 |
+
|
| 17 |
+
import re
|
| 18 |
+
import logging
|
| 19 |
+
from typing import Optional
|
| 20 |
+
|
| 21 |
+
logger = logging.getLogger(__name__)
|
| 22 |
+
|
| 23 |
+
# Lazy-loaded singletons — models are heavy, we only load them once and only when needed.
|
| 24 |
+
_category_pipeline = None
|
| 25 |
+
_sentiment_ar_pipeline = None
|
| 26 |
+
_sentiment_en_pipeline = None
|
| 27 |
+
_ner_ar_pipeline = None
|
| 28 |
+
_ner_en_pipeline = None
|
| 29 |
+
_keyword_model = None
|
| 30 |
+
_models_available = None # None = not checked yet, True/False after first attempt
|
| 31 |
+
|
| 32 |
+
|
| 33 |
+
# ---------------------------------------------------------------------------
|
| 34 |
+
# Our 13 categories (must match CATEGORIES_MAP keys in categories_lexicon.py)
|
| 35 |
+
# ---------------------------------------------------------------------------
|
| 36 |
+
CATEGORY_LABELS = [
|
| 37 |
+
"Sports", "Technology", "Gaming", "Business", "Health", "Science",
|
| 38 |
+
"Food", "Entertainment", "Politics", "Education", "Crime", "Weather", "Travel"
|
| 39 |
+
]
|
| 40 |
+
|
| 41 |
+
# ---------------------------------------------------------------------------
|
| 42 |
+
# Region inference from NER-detected locations
|
| 43 |
+
# ---------------------------------------------------------------------------
|
| 44 |
+
from services.lexicons import MIDDLE_EAST_COUNTRIES, LOCATIONS_MAP
|
| 45 |
+
|
| 46 |
+
|
| 47 |
+
def _check_dependencies() -> bool:
|
| 48 |
+
"""Check if transformers & torch are installed."""
|
| 49 |
+
global _models_available
|
| 50 |
+
if _models_available is not None:
|
| 51 |
+
return _models_available
|
| 52 |
+
try:
|
| 53 |
+
import transformers # noqa: F401
|
| 54 |
+
import torch # noqa: F401
|
| 55 |
+
_models_available = True
|
| 56 |
+
logger.info("NLP model dependencies (transformers, torch) are available.")
|
| 57 |
+
except ImportError as e:
|
| 58 |
+
_models_available = False
|
| 59 |
+
logger.warning(f"NLP model dependencies not installed ({e}). Tier 2 fallback disabled.")
|
| 60 |
+
return _models_available
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
# ========================= MODEL LOADERS =========================
|
| 64 |
+
|
| 65 |
+
def _get_category_pipeline():
|
| 66 |
+
"""Zero-shot classification using BART-MNLI. Handles Arabic & English."""
|
| 67 |
+
global _category_pipeline
|
| 68 |
+
if _category_pipeline is not None:
|
| 69 |
+
return _category_pipeline
|
| 70 |
+
try:
|
| 71 |
+
from transformers import pipeline
|
| 72 |
+
_category_pipeline = pipeline(
|
| 73 |
+
"zero-shot-classification",
|
| 74 |
+
model="facebook/bart-large-mnli",
|
| 75 |
+
device=-1 # CPU only (GitHub Actions has no GPU)
|
| 76 |
+
)
|
| 77 |
+
logger.info("Loaded zero-shot category classifier (facebook/bart-large-mnli)")
|
| 78 |
+
except Exception as e:
|
| 79 |
+
logger.warning(f"Failed to load category classifier: {e}")
|
| 80 |
+
_category_pipeline = None
|
| 81 |
+
return _category_pipeline
|
| 82 |
+
|
| 83 |
+
|
| 84 |
+
def _get_sentiment_pipeline(language: str):
|
| 85 |
+
"""Language-specific sentiment pipeline."""
|
| 86 |
+
global _sentiment_ar_pipeline, _sentiment_en_pipeline
|
| 87 |
+
|
| 88 |
+
if language == "ar":
|
| 89 |
+
if _sentiment_ar_pipeline is not None:
|
| 90 |
+
return _sentiment_ar_pipeline
|
| 91 |
+
try:
|
| 92 |
+
from transformers import pipeline
|
| 93 |
+
_sentiment_ar_pipeline = pipeline(
|
| 94 |
+
"text-classification",
|
| 95 |
+
model="CAMeL-Lab/bert-base-arabic-camelbert-msa-sentiment",
|
| 96 |
+
device=-1
|
| 97 |
+
)
|
| 98 |
+
logger.info("Loaded Arabic sentiment model (CAMeL-Lab/camelbert-msa-sentiment)")
|
| 99 |
+
except Exception as e:
|
| 100 |
+
logger.warning(f"Failed to load Arabic sentiment model: {e}")
|
| 101 |
+
_sentiment_ar_pipeline = None
|
| 102 |
+
return _sentiment_ar_pipeline
|
| 103 |
+
else:
|
| 104 |
+
if _sentiment_en_pipeline is not None:
|
| 105 |
+
return _sentiment_en_pipeline
|
| 106 |
+
try:
|
| 107 |
+
from transformers import pipeline
|
| 108 |
+
_sentiment_en_pipeline = pipeline(
|
| 109 |
+
"text-classification",
|
| 110 |
+
model="distilbert-base-uncased-finetuned-sst-2-english",
|
| 111 |
+
device=-1
|
| 112 |
+
)
|
| 113 |
+
logger.info("Loaded English sentiment model (distilbert-sst-2)")
|
| 114 |
+
except Exception as e:
|
| 115 |
+
logger.warning(f"Failed to load English sentiment model: {e}")
|
| 116 |
+
_sentiment_en_pipeline = None
|
| 117 |
+
return _sentiment_en_pipeline
|
| 118 |
+
|
| 119 |
+
|
| 120 |
+
def _get_ner_pipeline(language: str):
|
| 121 |
+
"""Language-specific NER pipeline."""
|
| 122 |
+
global _ner_ar_pipeline, _ner_en_pipeline
|
| 123 |
+
|
| 124 |
+
if language == "ar":
|
| 125 |
+
if _ner_ar_pipeline is not None:
|
| 126 |
+
return _ner_ar_pipeline
|
| 127 |
+
try:
|
| 128 |
+
from transformers import pipeline
|
| 129 |
+
_ner_ar_pipeline = pipeline(
|
| 130 |
+
"ner",
|
| 131 |
+
model="CAMeL-Lab/bert-base-arabic-camelbert-msa-ner",
|
| 132 |
+
aggregation_strategy="simple",
|
| 133 |
+
device=-1
|
| 134 |
+
)
|
| 135 |
+
logger.info("Loaded Arabic NER model (CAMeL-Lab/camelbert-msa-ner)")
|
| 136 |
+
except Exception as e:
|
| 137 |
+
logger.warning(f"Failed to load Arabic NER model: {e}")
|
| 138 |
+
_ner_ar_pipeline = None
|
| 139 |
+
return _ner_ar_pipeline
|
| 140 |
+
else:
|
| 141 |
+
if _ner_en_pipeline is not None:
|
| 142 |
+
return _ner_en_pipeline
|
| 143 |
+
try:
|
| 144 |
+
from transformers import pipeline
|
| 145 |
+
_ner_en_pipeline = pipeline(
|
| 146 |
+
"ner",
|
| 147 |
+
model="dslim/bert-base-NER",
|
| 148 |
+
aggregation_strategy="simple",
|
| 149 |
+
device=-1
|
| 150 |
+
)
|
| 151 |
+
logger.info("Loaded English NER model (dslim/bert-base-NER)")
|
| 152 |
+
except Exception as e:
|
| 153 |
+
logger.warning(f"Failed to load English NER model: {e}")
|
| 154 |
+
_ner_en_pipeline = None
|
| 155 |
+
return _ner_en_pipeline
|
| 156 |
+
|
| 157 |
+
|
| 158 |
+
def _get_keyword_model():
|
| 159 |
+
"""KeyBERT with multilingual sentence-transformer embeddings."""
|
| 160 |
+
global _keyword_model
|
| 161 |
+
if _keyword_model is not None:
|
| 162 |
+
return _keyword_model
|
| 163 |
+
try:
|
| 164 |
+
from keybert import KeyBERT
|
| 165 |
+
_keyword_model = KeyBERT(model="paraphrase-multilingual-MiniLM-L12-v2")
|
| 166 |
+
logger.info("Loaded KeyBERT keyword extractor (multilingual-MiniLM)")
|
| 167 |
+
except Exception as e:
|
| 168 |
+
logger.warning(f"Failed to load KeyBERT: {e}")
|
| 169 |
+
_keyword_model = None
|
| 170 |
+
return _keyword_model
|
| 171 |
+
|
| 172 |
+
|
| 173 |
+
# ========================= ENRICHMENT =========================
|
| 174 |
+
|
| 175 |
+
class LocalNLPClassifier:
|
| 176 |
+
"""
|
| 177 |
+
Tier 2 NLP enrichment engine using pre-trained transformer models.
|
| 178 |
+
|
| 179 |
+
Usage:
|
| 180 |
+
classifier = LocalNLPClassifier()
|
| 181 |
+
if classifier.is_available():
|
| 182 |
+
enriched_article = classifier.enrich_article(article)
|
| 183 |
+
"""
|
| 184 |
+
|
| 185 |
+
def is_available(self) -> bool:
|
| 186 |
+
"""Check if the required ML libraries are installed."""
|
| 187 |
+
return _check_dependencies()
|
| 188 |
+
|
| 189 |
+
def enrich_article(self, article) -> "Article":
|
| 190 |
+
"""
|
| 191 |
+
Enrich a single article using local transformer models.
|
| 192 |
+
Each sub-task is wrapped in try/except so a failure in one
|
| 193 |
+
(e.g. sentiment) does not block the others (e.g. NER).
|
| 194 |
+
"""
|
| 195 |
+
text = f"{article.title} {article.summary or ''}"
|
| 196 |
+
lang = getattr(article, 'language', 'en')
|
| 197 |
+
|
| 198 |
+
# 1. Category (zero-shot classification)
|
| 199 |
+
try:
|
| 200 |
+
self._classify_category(article, text)
|
| 201 |
+
except Exception as e:
|
| 202 |
+
logger.warning(f"Tier 2 category classification failed: {e}")
|
| 203 |
+
|
| 204 |
+
# 2. Sentiment
|
| 205 |
+
try:
|
| 206 |
+
self._classify_sentiment(article, text, lang)
|
| 207 |
+
except Exception as e:
|
| 208 |
+
logger.warning(f"Tier 2 sentiment analysis failed: {e}")
|
| 209 |
+
|
| 210 |
+
# 3. Named Entity Recognition -> people, organizations, locations, country, region
|
| 211 |
+
try:
|
| 212 |
+
self._extract_entities(article, text, lang)
|
| 213 |
+
except Exception as e:
|
| 214 |
+
logger.warning(f"Tier 2 NER failed: {e}")
|
| 215 |
+
|
| 216 |
+
# 4. Keywords & Tags
|
| 217 |
+
try:
|
| 218 |
+
self._extract_keywords(article, text)
|
| 219 |
+
except Exception as e:
|
| 220 |
+
logger.warning(f"Tier 2 keyword extraction failed: {e}")
|
| 221 |
+
|
| 222 |
+
# 5. Importance scoring (derived from NER entity density + category)
|
| 223 |
+
try:
|
| 224 |
+
self._score_importance(article, text)
|
| 225 |
+
except Exception as e:
|
| 226 |
+
logger.warning(f"Tier 2 importance scoring failed: {e}")
|
| 227 |
+
|
| 228 |
+
# 6. Reading time estimation
|
| 229 |
+
if not article.readingTime:
|
| 230 |
+
article.readingTime = self._estimate_reading_time(text)
|
| 231 |
+
|
| 232 |
+
return article
|
| 233 |
+
|
| 234 |
+
# ---------------------------------------------------------------
|
| 235 |
+
# Sub-task implementations
|
| 236 |
+
# ---------------------------------------------------------------
|
| 237 |
+
|
| 238 |
+
def _classify_category(self, article, text: str):
|
| 239 |
+
"""Zero-shot classification into our 13 news categories."""
|
| 240 |
+
if article.category and article.category not in ("General", "Middle East", "Egypt", "World", "North Africa"):
|
| 241 |
+
return # Already classified by the RSS source metadata
|
| 242 |
+
|
| 243 |
+
pipe = _get_category_pipeline()
|
| 244 |
+
if pipe is None:
|
| 245 |
+
return
|
| 246 |
+
|
| 247 |
+
# Truncate to 512 tokens for BART
|
| 248 |
+
result = pipe(text[:1024], candidate_labels=CATEGORY_LABELS, multi_label=False)
|
| 249 |
+
if result and result['scores'][0] > 0.25:
|
| 250 |
+
article.category = result['labels'][0]
|
| 251 |
+
# Subcategory: use the second-highest label if confident
|
| 252 |
+
if len(result['labels']) > 1 and result['scores'][1] > 0.20:
|
| 253 |
+
article.subcategory = result['labels'][1]
|
| 254 |
+
else:
|
| 255 |
+
article.category = article.category or "General"
|
| 256 |
+
|
| 257 |
+
def _classify_sentiment(self, article, text: str, lang: str):
|
| 258 |
+
"""Sentiment analysis using language-specific pre-trained models."""
|
| 259 |
+
pipe = _get_sentiment_pipeline(lang)
|
| 260 |
+
if pipe is None:
|
| 261 |
+
return
|
| 262 |
+
|
| 263 |
+
result = pipe(text[:512])
|
| 264 |
+
if not result:
|
| 265 |
+
return
|
| 266 |
+
|
| 267 |
+
label = result[0]['label'].upper()
|
| 268 |
+
score = result[0]['score']
|
| 269 |
+
|
| 270 |
+
if lang == "ar":
|
| 271 |
+
# CAMeLBERT sentiment outputs: positive, negative, neutral
|
| 272 |
+
label_map = {"POSITIVE": "Positive", "NEGATIVE": "Negative", "NEUTRAL": "Neutral"}
|
| 273 |
+
article.sentiment = label_map.get(label, "Neutral")
|
| 274 |
+
else:
|
| 275 |
+
# DistilBERT SST-2 outputs: POSITIVE, NEGATIVE
|
| 276 |
+
if label == "POSITIVE":
|
| 277 |
+
article.sentiment = "Positive"
|
| 278 |
+
elif label == "NEGATIVE":
|
| 279 |
+
article.sentiment = "Negative"
|
| 280 |
+
else:
|
| 281 |
+
article.sentiment = "Neutral"
|
| 282 |
+
|
| 283 |
+
# Mixed sentiment: if confidence is low (close to 50/50)
|
| 284 |
+
if 0.45 < score < 0.60:
|
| 285 |
+
article.sentiment = "Mixed"
|
| 286 |
+
|
| 287 |
+
def _extract_entities(self, article, text: str, lang: str):
|
| 288 |
+
"""NER extraction: people, organizations, locations."""
|
| 289 |
+
pipe = _get_ner_pipeline(lang)
|
| 290 |
+
if pipe is None:
|
| 291 |
+
return
|
| 292 |
+
|
| 293 |
+
entities = pipe(text[:512])
|
| 294 |
+
if not entities:
|
| 295 |
+
return
|
| 296 |
+
|
| 297 |
+
people = []
|
| 298 |
+
organizations = []
|
| 299 |
+
locations = []
|
| 300 |
+
|
| 301 |
+
for ent in entities:
|
| 302 |
+
entity_group = ent.get('entity_group', '')
|
| 303 |
+
word = ent.get('word', '').strip()
|
| 304 |
+
score = ent.get('score', 0)
|
| 305 |
+
|
| 306 |
+
# Filter low-confidence entities
|
| 307 |
+
if score < 0.5 or len(word) < 2:
|
| 308 |
+
continue
|
| 309 |
+
|
| 310 |
+
# Clean up subword tokens (##prefix)
|
| 311 |
+
word = word.replace('##', '')
|
| 312 |
+
|
| 313 |
+
if entity_group in ('PER', 'B-PER', 'I-PER'):
|
| 314 |
+
if word not in people:
|
| 315 |
+
people.append(word)
|
| 316 |
+
elif entity_group in ('ORG', 'B-ORG', 'I-ORG'):
|
| 317 |
+
if word not in organizations:
|
| 318 |
+
organizations.append(word)
|
| 319 |
+
elif entity_group in ('LOC', 'B-LOC', 'I-LOC', 'GPE', 'B-GPE', 'I-GPE'):
|
| 320 |
+
if word not in locations:
|
| 321 |
+
locations.append(word)
|
| 322 |
+
|
| 323 |
+
article.people = people[:10] # Cap at 10
|
| 324 |
+
article.organizations = organizations[:10]
|
| 325 |
+
article.locations = locations[:10] if locations else article.locations
|
| 326 |
+
|
| 327 |
+
# Derive country and region from detected locations
|
| 328 |
+
if locations:
|
| 329 |
+
# Try to map NER location names to our gazetteer
|
| 330 |
+
for loc in locations:
|
| 331 |
+
loc_lower = loc.lower()
|
| 332 |
+
if loc_lower in LOCATIONS_MAP:
|
| 333 |
+
article.country = LOCATIONS_MAP[loc_lower]
|
| 334 |
+
article.region = "Middle East" if article.country in MIDDLE_EAST_COUNTRIES else "Global"
|
| 335 |
+
break
|
| 336 |
+
if not article.country:
|
| 337 |
+
article.country = locations[0]
|
| 338 |
+
article.region = "Global"
|
| 339 |
+
|
| 340 |
+
def _extract_keywords(self, article, text: str):
|
| 341 |
+
"""KeyBERT keyword extraction with multilingual embeddings."""
|
| 342 |
+
kw_model = _get_keyword_model()
|
| 343 |
+
if kw_model is None:
|
| 344 |
+
return
|
| 345 |
+
|
| 346 |
+
try:
|
| 347 |
+
keywords = kw_model.extract_keywords(
|
| 348 |
+
text,
|
| 349 |
+
keyphrase_ngram_range=(1, 2),
|
| 350 |
+
stop_words=None, # KeyBERT handles multilingual stop words
|
| 351 |
+
top_n=12,
|
| 352 |
+
use_maxsum=True,
|
| 353 |
+
nr_candidates=20
|
| 354 |
+
)
|
| 355 |
+
except Exception:
|
| 356 |
+
return
|
| 357 |
+
|
| 358 |
+
if keywords:
|
| 359 |
+
article.keywords = [kw[0] for kw in keywords[:6]]
|
| 360 |
+
article.tags = [kw[0] for kw in keywords[:10]]
|
| 361 |
+
|
| 362 |
+
def _score_importance(self, article, text: str):
|
| 363 |
+
"""
|
| 364 |
+
Importance scoring (1-10) based on:
|
| 365 |
+
- Number of named entities (more entities = more important)
|
| 366 |
+
- High-impact trigger words
|
| 367 |
+
- Category weight
|
| 368 |
+
"""
|
| 369 |
+
score = 5 # Base importance
|
| 370 |
+
|
| 371 |
+
# Named entity density bonus
|
| 372 |
+
entity_count = len(article.people or []) + len(article.organizations or []) + len(article.locations or [])
|
| 373 |
+
if entity_count >= 5:
|
| 374 |
+
score += 2
|
| 375 |
+
elif entity_count >= 2:
|
| 376 |
+
score += 1
|
| 377 |
+
|
| 378 |
+
# High-impact trigger words
|
| 379 |
+
high_impact_terms = [
|
| 380 |
+
"عاجل", "قمة", "حرب", "وفاة", "رئيس", "كوارث", "اغتيال", "انقلاب", "زلزال",
|
| 381 |
+
"breaking", "war", "president", "disaster", "crisis", "urgent", "killed",
|
| 382 |
+
"earthquake", "assassination", "coup", "summit", "death"
|
| 383 |
+
]
|
| 384 |
+
text_lower = text.lower()
|
| 385 |
+
if any(term in text_lower for term in high_impact_terms):
|
| 386 |
+
score += 3
|
| 387 |
+
|
| 388 |
+
# Category bonus
|
| 389 |
+
high_importance_categories = {"Politics", "Crime", "Health", "Science"}
|
| 390 |
+
if article.category in high_importance_categories:
|
| 391 |
+
score += 1
|
| 392 |
+
|
| 393 |
+
article.importance = min(10, max(1, score))
|
| 394 |
+
|
| 395 |
+
def _estimate_reading_time(self, text: str) -> int:
|
| 396 |
+
"""Estimate reading time in minutes from word count."""
|
| 397 |
+
clean_text = re.sub(r'<[^>]+>', '', text)
|
| 398 |
+
words = len(clean_text.split())
|
| 399 |
+
return max(1, round(words / 180))
|