JsonDONSON commited on
Commit
6545796
·
verified ·
1 Parent(s): d0f8046

Upload folder using huggingface_hub

Browse files
Files changed (50) hide show
  1. .idea/.gitignore +3 -0
  2. .idea/WhatsNew-Scraper.iml +9 -0
  3. .idea/caches/deviceStreaming.xml +0 -0
  4. .idea/misc.xml +19 -0
  5. .idea/modules.xml +8 -0
  6. .idea/studiobot.xml +6 -0
  7. .idea/vcs.xml +6 -0
  8. .idea/workspace.xml +55 -0
  9. __pycache__/main.cpython-314.pyc +0 -0
  10. app.py +269 -0
  11. data/__pycache__/firestore_repo.cpython-314.pyc +0 -0
  12. data/repositories/__pycache__/article_repository_impl.cpython-314.pyc +0 -0
  13. data/repositories/article_repository_impl.py +58 -0
  14. domain/__pycache__/interfaces.cpython-314.pyc +0 -0
  15. domain/__pycache__/models.cpython-314.pyc +0 -0
  16. domain/entities/__pycache__/article.cpython-314.pyc +0 -0
  17. domain/entities/article.py +74 -0
  18. domain/interfaces.py +10 -0
  19. domain/repositories/__pycache__/article_repository.cpython-314.pyc +0 -0
  20. domain/repositories/article_repository.py +14 -0
  21. domain/usecases/__pycache__/scrape_and_enrich.cpython-314.pyc +0 -0
  22. domain/usecases/scrape_and_enrich.py +69 -0
  23. main.py +133 -0
  24. nlp_training/__init__.py +1 -0
  25. nlp_training/prepare_datasets.py +398 -0
  26. nlp_training/train_category_classifier.py +190 -0
  27. nlp_training/train_ner_model.py +271 -0
  28. nlp_training/train_sentiment_analyzer.py +195 -0
  29. requirements.txt +54 -0
  30. requirements_space.txt +9 -0
  31. scrapers/__pycache__/rss_scraper.cpython-314.pyc +0 -0
  32. scrapers/base_scraper.py +21 -0
  33. scrapers/rss_scraper.py +128 -0
  34. services/__pycache__/ai_classifier.cpython-314.pyc +0 -0
  35. services/__pycache__/duplicate_detector.cpython-314.pyc +0 -0
  36. services/ai_classifier.py +185 -0
  37. services/duplicate_detector.py +45 -0
  38. services/lexicons/__init__.py +15 -0
  39. services/lexicons/__pycache__/__init__.cpython-314.pyc +0 -0
  40. services/lexicons/__pycache__/categories_lexicon.cpython-314.pyc +0 -0
  41. services/lexicons/__pycache__/images_lexicon.cpython-314.pyc +0 -0
  42. services/lexicons/__pycache__/locations_lexicon.cpython-314.pyc +0 -0
  43. services/lexicons/__pycache__/sentiment_lexicon.cpython-314.pyc +0 -0
  44. services/lexicons/__pycache__/stopwords_lexicon.cpython-314.pyc +0 -0
  45. services/lexicons/categories_lexicon.py +135 -0
  46. services/lexicons/images_lexicon.py +21 -0
  47. services/lexicons/locations_lexicon.py +59 -0
  48. services/lexicons/sentiment_lexicon.py +46 -0
  49. services/lexicons/stopwords_lexicon.py +20 -0
  50. services/local_nlp_classifier.py +399 -0
.idea/.gitignore ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ # Default ignored files
2
+ /shelf/
3
+ /workspace.xml
.idea/WhatsNew-Scraper.iml ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ <?xml version="1.0" encoding="UTF-8"?>
2
+ <module type="JAVA_MODULE" version="4">
3
+ <component name="NewModuleRootManager" inherit-compiler-output="true">
4
+ <exclude-output />
5
+ <content url="file://$MODULE_DIR$" />
6
+ <orderEntry type="inheritedJdk" />
7
+ <orderEntry type="sourceFolder" forTests="false" />
8
+ </component>
9
+ </module>
.idea/caches/deviceStreaming.xml ADDED
The diff for this file is too large to render. See raw diff
 
.idea/misc.xml ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ <?xml version="1.0" encoding="UTF-8"?>
2
+ <project version="4">
3
+ <component name="ProjectRootManager" version="2" languageLevel="JDK_21" default="true" project-jdk-name="jbr-21" project-jdk-type="JavaSDK">
4
+ <output url="file://$PROJECT_DIR$/out" />
5
+ </component>
6
+ <component name="StoredSessionsInfoStore">
7
+ <option name="sessions">
8
+ <list>
9
+ <SessionInfoState>
10
+ <option name="id" value="20260805-000758-9a7987ea-2159-4416-abc2-15f38fb7cc6d" />
11
+ <option name="lastUpdateTimestamp" value="1785877678080" />
12
+ <option name="responseMode" value="Agent" />
13
+ <option name="title" value="New Agent Session" />
14
+ <option name="user" value="abdo1234567es@gmail.com" />
15
+ </SessionInfoState>
16
+ </list>
17
+ </option>
18
+ </component>
19
+ </project>
.idea/modules.xml ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ <?xml version="1.0" encoding="UTF-8"?>
2
+ <project version="4">
3
+ <component name="ProjectModuleManager">
4
+ <modules>
5
+ <module fileurl="file://$PROJECT_DIR$/.idea/WhatsNew-Scraper.iml" filepath="$PROJECT_DIR$/.idea/WhatsNew-Scraper.iml" />
6
+ </modules>
7
+ </component>
8
+ </project>
.idea/studiobot.xml ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ <?xml version="1.0" encoding="UTF-8"?>
2
+ <project version="4">
3
+ <component name="StudioBotProjectSettings">
4
+ <option name="shareContext" value="OptedIn" />
5
+ </component>
6
+ </project>
.idea/vcs.xml ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ <?xml version="1.0" encoding="UTF-8"?>
2
+ <project version="4">
3
+ <component name="VcsDirectoryMappings">
4
+ <mapping directory="" vcs="Git" />
5
+ </component>
6
+ </project>
.idea/workspace.xml ADDED
@@ -0,0 +1,55 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ <?xml version="1.0" encoding="UTF-8"?>
2
+ <project version="4">
3
+ <component name="AutoImportSettings">
4
+ <option name="autoReloadType" value="NONE" />
5
+ </component>
6
+ <component name="ChangeListManager">
7
+ <list default="true" id="66ee3a4c-7587-42f2-b266-bea278ef11ef" name="Changes" comment="" />
8
+ <option name="SHOW_DIALOG" value="false" />
9
+ <option name="HIGHLIGHT_CONFLICTS" value="true" />
10
+ <option name="HIGHLIGHT_NON_ACTIVE_CHANGELIST" value="false" />
11
+ <option name="LAST_RESOLUTION" value="IGNORE" />
12
+ </component>
13
+ <component name="ClangdSettings">
14
+ <option name="formatViaClangd" value="false" />
15
+ </component>
16
+ <component name="Git.Settings">
17
+ <option name="RECENT_GIT_ROOT_PATH" value="$PROJECT_DIR$" />
18
+ </component>
19
+ <component name="ProjectColorInfo"><![CDATA[{
20
+ "associatedIndex": 2,
21
+ "fromUser": false
22
+ }]]></component>
23
+ <component name="ProjectId" id="3HdoQC2fewtmSXjlJPtirHTWhuF" />
24
+ <component name="ProjectViewState">
25
+ <option name="hideEmptyMiddlePackages" value="true" />
26
+ <option name="showLibraryContents" value="true" />
27
+ </component>
28
+ <component name="PropertiesComponent"><![CDATA[{
29
+ "keyToString": {
30
+ "ModuleVcsDetector.initialDetectionPerformed": "true",
31
+ "RunOnceActivity.ShowReadmeOnStart": "true",
32
+ "RunOnceActivity.cidr.known.project.marker": "true",
33
+ "RunOnceActivity.git.unshallow": "true",
34
+ "RunOnceActivity.readMode.enableVisualFormatting": "true",
35
+ "cf.first.check.clang-format": "false",
36
+ "cidr.known.project.marker": "true",
37
+ "dart.analysis.tool.window.visible": "false",
38
+ "git-widget-placeholder": "main",
39
+ "kotlin-language-version-configured": "true",
40
+ "last_opened_file_path": "C:/Users/Abdo/AndroidStudioProjects/WhatsNew-Scraper",
41
+ "settings.editor.selected.configurable": "preferences.lookFeel",
42
+ "show.migrate.to.gradle.popup": "false"
43
+ }
44
+ }]]></component>
45
+ <component name="TaskManager">
46
+ <task active="true" id="Default" summary="Default task">
47
+ <changelist id="66ee3a4c-7587-42f2-b266-bea278ef11ef" name="Changes" comment="" />
48
+ <created>1786207287629</created>
49
+ <option name="number" value="Default" />
50
+ <option name="presentableId" value="Default" />
51
+ <updated>1786207287629</updated>
52
+ </task>
53
+ <servers />
54
+ </component>
55
+ </project>
__pycache__/main.cpython-314.pyc ADDED
Binary file (8.17 kB). View file
 
app.py ADDED
@@ -0,0 +1,269 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ WhatsNew News Intelligence Engine — Hugging Face Gradio Web Space.
3
+
4
+ Paste a news article URL (Arabic or English) or paste article text directly
5
+ to perform real-time NLP enrichment: Category, Sentiment, NER, Keywords,
6
+ Geocoding, and Metadata.
7
+ """
8
+
9
+ import os
10
+ import re
11
+ import json
12
+ import logging
13
+ import requests
14
+ from bs4 import BeautifulSoup
15
+ import gradio as gr
16
+
17
+ from domain.entities.article import Article
18
+ from services.local_nlp_classifier import LocalNLPClassifier
19
+
20
+ logging.basicConfig(level=logging.INFO)
21
+ logger = logging.getLogger(__name__)
22
+
23
+ # Initialize local NLP classifier engine
24
+ nlp_engine = LocalNLPClassifier()
25
+
26
+
27
+ def scrape_article_from_url(url: str):
28
+ """Scrape article title, summary, and image from a news URL."""
29
+ headers = {
30
+ "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
31
+ }
32
+ try:
33
+ response = requests.get(url.strip(), headers=headers, timeout=8)
34
+ if response.status_code != 200:
35
+ return None, None, None, f"Failed to fetch URL (Status {response.status_code})"
36
+
37
+ soup = BeautifulSoup(response.text, 'html.parser')
38
+
39
+ # 1. Title Extraction
40
+ title = None
41
+ og_title = soup.find('meta', property='og:title')
42
+ if og_title and og_title.get('content'):
43
+ title = og_title['content'].strip()
44
+ elif soup.title and soup.title.string:
45
+ title = soup.title.string.strip()
46
+ elif soup.find('h1'):
47
+ title = soup.find('h1').get_text().strip()
48
+
49
+ # 2. Image Extraction
50
+ image_url = None
51
+ og_image = soup.find('meta', property='og:image') or soup.find('meta', attrs={'name': 'twitter:image'})
52
+ if og_image and og_image.get('content'):
53
+ image_url = og_image['content'].strip()
54
+
55
+ # 3. Summary / Body Extraction
56
+ summary = None
57
+ og_desc = soup.find('meta', property='og:description') or soup.find('meta', attrs={'name': 'description'})
58
+ if og_desc and og_desc.get('content'):
59
+ summary = og_desc['content'].strip()
60
+ else:
61
+ paragraphs = [p.get_text().strip() for p in soup.find_all('p') if len(p.get_text().strip()) > 30]
62
+ summary = ' '.join(paragraphs[:3]) if paragraphs else ""
63
+
64
+ return title, summary, image_url, None
65
+ except Exception as e:
66
+ return None, None, None, f"Error scraping URL: {str(e)}"
67
+
68
+
69
+ def process_nlp_analysis(url_input: str, title_input: str, summary_input: str):
70
+ """Main Gradio processing handler."""
71
+ url = (url_input or "").strip()
72
+ title = (title_input or "").strip()
73
+ summary = (summary_input or "").strip()
74
+ image_url = None
75
+ status_msg = ""
76
+
77
+ # If URL provided, attempt web scraping first
78
+ if url:
79
+ scraped_title, scraped_summary, scraped_img, err = scrape_article_from_url(url)
80
+ if err:
81
+ status_msg = f"⚠️ {err}. Using manual text inputs."
82
+ else:
83
+ title = scraped_title or title
84
+ summary = scraped_summary or summary
85
+ image_url = scraped_img
86
+ status_msg = "✅ Successfully scraped article from live URL!"
87
+
88
+ if not title:
89
+ return (
90
+ "⚠️ Please enter an article URL or provide a Title & Summary.",
91
+ "", "", "", "", "", "", "", "", "", {}
92
+ )
93
+
94
+ # Detect language
95
+ is_arabic = any('\u0600' <= char <= '\u06FF' for char in f"{title} {summary}")
96
+ language = "ar" if is_arabic else "en"
97
+
98
+ # Construct Article entity
99
+ article = Article(
100
+ title=title,
101
+ summary=summary or title,
102
+ url=url or "https://whatsnew.ai/demo",
103
+ source="Gradio Interactive UI",
104
+ publishedAt=None,
105
+ language=language,
106
+ imageUrl=image_url
107
+ )
108
+
109
+ # Perform NLP enrichment
110
+ enriched = nlp_engine.enrich_article(article)
111
+
112
+ # Formatted Visual Displays
113
+ category_display = f"🏷️ **Category**: {enriched.category or 'General'}\n📌 **Subcategory**: {enriched.subcategory or 'N/A'}"
114
+
115
+ # Sentiment Badge
116
+ sent = (enriched.sentiment or "Neutral").title()
117
+ sent_icon = "🟢" if sent == "Positive" else "🔴" if sent == "Negative" else "⚪"
118
+ sentiment_display = f"{sent_icon} **Sentiment**: {sent}"
119
+
120
+ # Named Entities
121
+ people_str = ", ".join(enriched.people) if enriched.people else "None detected"
122
+ orgs_str = ", ".join(enriched.organizations) if enriched.organizations else "None detected"
123
+ locs_str = ", ".join(enriched.locations) if enriched.locations else "None detected"
124
+ ner_display = f"👤 **People**: {people_str}\n🏢 **Organizations**: {orgs_str}\n📍 **Locations**: {locs_str}"
125
+
126
+ # Geocoding
127
+ geo_display = f"🌍 **Region**: {enriched.region or 'Global'}\n🏳️ **Country**: {enriched.country or 'Global'}"
128
+
129
+ # Tags & Keywords
130
+ tags_display = " ".join([f"#{t.replace(' ', '_')}" for t in (enriched.tags or [])])
131
+ keywords_display = ", ".join(enriched.keywords or [])
132
+
133
+ # Stats
134
+ stats_display = f"⭐ **Importance Score**: {enriched.importance or 5}/10\n⏱️ **Reading Time**: {enriched.readingTime or 1} min\n🌐 **Language**: {'Arabic 🇦🇪/🇪🇬' if language == 'ar' else 'English 🇬🇧/🇺🇸'}"
135
+
136
+ # Full JSON Metadata dictionary
137
+ json_output = {
138
+ "title": enriched.title,
139
+ "summary": enriched.summary,
140
+ "url": enriched.url,
141
+ "imageUrl": enriched.imageUrl,
142
+ "language": enriched.language,
143
+ "category": enriched.category,
144
+ "subcategory": enriched.subcategory,
145
+ "sentiment": enriched.sentiment,
146
+ "region": enriched.region,
147
+ "country": enriched.country,
148
+ "people": enriched.people,
149
+ "organizations": enriched.organizations,
150
+ "locations": enriched.locations,
151
+ "tags": enriched.tags,
152
+ "keywords": enriched.keywords,
153
+ "importance": enriched.importance,
154
+ "readingTime": enriched.readingTime
155
+ }
156
+
157
+ return (
158
+ status_msg or "✅ Analysis Complete",
159
+ title,
160
+ summary,
161
+ category_display,
162
+ sentiment_display,
163
+ ner_display,
164
+ geo_display,
165
+ tags_display,
166
+ keywords_display,
167
+ stats_display,
168
+ json_output,
169
+ image_url or "https://images.unsplash.com/photo-1504711434969-e33886168f5c?q=80&w=1000&auto=format&fit=crop"
170
+ )
171
+
172
+
173
+ # ---------------------------------------------------------------------------
174
+ # Custom CSS Aesthetics
175
+ # ---------------------------------------------------------------------------
176
+ custom_css = """
177
+ .main-title { text-align: center; font-size: 2.2em; font-weight: bold; margin-bottom: 5px; color: #1E293B; }
178
+ .subtitle { text-align: center; font-size: 1.1em; color: #64748B; margin-bottom: 25px; }
179
+ .output-card { border-radius: 12px; padding: 15px; background: #F8FAFC; border: 1px solid #E2E8F0; }
180
+ """
181
+
182
+ # ---------------------------------------------------------------------------
183
+ # Gradio UI Layout
184
+ # ---------------------------------------------------------------------------
185
+ with gr.Blocks(title="WhatsNew NLP Engine", css=custom_css, theme=gr.themes.Soft()) as demo:
186
+ gr.Markdown("<div class='main-title'>📰 WhatsNew Intelligence Engine</div>")
187
+ gr.Markdown("<div class='subtitle'>Real-Time Multilingual NLP Article Categorization, Sentiment Analysis, NER & Metadata Extraction</div>")
188
+
189
+ with gr.Row():
190
+ with gr.Column(scale=1):
191
+ gr.Markdown("### 📥 Article Input")
192
+ url_input = gr.Textbox(
193
+ label="Option A: Paste News Article URL (Arabic or English)",
194
+ placeholder="https://www.aljazeera.net/... or https://techcrunch.com/...",
195
+ lines=1
196
+ )
197
+ gr.Markdown("**OR Enter Manually:**")
198
+ title_input = gr.Textbox(
199
+ label="Article Title",
200
+ placeholder="e.g. الأهلي يفوز بلقب دوري أبطال أفريقيا",
201
+ lines=2
202
+ )
203
+ summary_input = gr.Textbox(
204
+ label="Article Summary / Text",
205
+ placeholder="Paste news paragraph or text summary here...",
206
+ lines=4
207
+ )
208
+ analyze_btn = gr.Button("🚀 Analyze & Enrich Article", variant="primary", size="lg")
209
+
210
+ gr.Markdown("---")
211
+ gr.Markdown("### 💡 Quick Examples")
212
+ gr.Examples(
213
+ examples=[
214
+ ["", "الأهلي يتوج بكأس السوبر المصري بعد فوزه على الزمالك في ركلات الترجيح", "نجح الفريق الأول لكرة القدم بنادي الأهلي في تحقيق حسم بطولة كأس السوبر المصري بعد مباراة حماسية انتهت بفوزه على غريمه التقليدي الزمالك."],
215
+ ["", "Apple unveils new M4 MacBook Pro with advanced AI capabilities", "Apple today announced its next-generation MacBook Pro powered by the M4 chip lineup, featuring enhanced neural processing for on-device artificial intelligence."],
216
+ ["", "ارتفاع أسعار النفط العالمية بسبب التوترات الجيوسياسية في الشرق الأوسط", "شهدت أسواق النفط العالمية ارتفاعاً ملحوظاً في أسعار الخام وسط متابعة حثيثة من المستثمرين لتطورات الأوضاع الاقتصادية."]
217
+ ],
218
+ inputs=[url_input, title_input, summary_input]
219
+ )
220
+
221
+ with gr.Column(scale=1):
222
+ status_box = gr.Markdown("Waiting for input...")
223
+
224
+ with gr.Tabs():
225
+ with gr.TabItem("📊 NLP Insights"):
226
+ with gr.Row():
227
+ category_box = gr.Markdown(elem_classes=["output-card"])
228
+ sentiment_box = gr.Markdown(elem_classes=["output-card"])
229
+
230
+ ner_box = gr.Markdown(elem_classes=["output-card"])
231
+ geo_box = gr.Markdown(elem_classes=["output-card"])
232
+ stats_box = gr.Markdown(elem_classes=["output-card"])
233
+
234
+ gr.Markdown("#### 🏷️ Extracted Hashtags & Tags")
235
+ tags_box = gr.Markdown()
236
+
237
+ gr.Markdown("#### 🔑 Key Phrases")
238
+ keywords_box = gr.Markdown()
239
+
240
+ with gr.TabItem("🖼️ Media & Preview"):
241
+ image_preview = gr.Image(label="Article Cover Image", type="url")
242
+ extracted_title = gr.Textbox(label="Processed Title", interactive=False)
243
+ extracted_summary = gr.Textbox(label="Processed Summary", interactive=False, lines=4)
244
+
245
+ with gr.TabItem("📄 Raw JSON Metadata"):
246
+ json_box = gr.JSON(label="Full Firestore-Ready Document Schema")
247
+
248
+ # Connect click event
249
+ analyze_btn.click(
250
+ fn=process_nlp_analysis,
251
+ inputs=[url_input, title_input, summary_input],
252
+ outputs=[
253
+ status_box,
254
+ extracted_title,
255
+ extracted_summary,
256
+ category_box,
257
+ sentiment_box,
258
+ ner_box,
259
+ geo_box,
260
+ tags_box,
261
+ keywords_box,
262
+ stats_box,
263
+ json_box,
264
+ image_preview
265
+ ]
266
+ )
267
+
268
+ if __name__ == "__main__":
269
+ demo.launch()
data/__pycache__/firestore_repo.cpython-314.pyc ADDED
Binary file (3.13 kB). View file
 
data/repositories/__pycache__/article_repository_impl.cpython-314.pyc ADDED
Binary file (3.95 kB). View file
 
data/repositories/article_repository_impl.py ADDED
@@ -0,0 +1,58 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import firebase_admin
2
+ from firebase_admin import credentials, firestore
3
+ from typing import List
4
+ import logging
5
+ from domain.entities.article import Article
6
+ from domain.repositories.article_repository import ArticleRepository
7
+
8
+ logger = logging.getLogger(__name__)
9
+
10
+ class FirestoreArticleRepository(ArticleRepository):
11
+ def __init__(self, service_account_path: str = "service-account.json"):
12
+ try:
13
+ if not firebase_admin._apps:
14
+ cred = credentials.Certificate(service_account_path)
15
+ firebase_admin.initialize_app(cred)
16
+ self.db = firestore.client()
17
+ except Exception as e:
18
+ logger.error(f"Failed to initialize Firebase Admin: {e}")
19
+ self.db = None
20
+
21
+ def save_articles(self, articles: List[Article]) -> List[Article]:
22
+ if not self.db:
23
+ logger.error("Firestore DB not initialized. Cannot save articles.")
24
+ return []
25
+
26
+ articles_ref = self.db.collection('articles')
27
+ new_articles = []
28
+ for article in articles:
29
+ try:
30
+ # Use the generated deterministic ID as the document ID
31
+ doc_id = article.id
32
+ doc_ref = articles_ref.document(doc_id)
33
+
34
+ # Check if document exists
35
+ doc_snap = doc_ref.get()
36
+ if not doc_snap.exists:
37
+ doc_ref.set(article.model_dump(mode='json'))
38
+ new_articles.append(article)
39
+ else:
40
+ # Backfill/Update image if existing document was saved without an image
41
+ existing_data = doc_snap.to_dict() or {}
42
+ if not existing_data.get('imageUrl') and article.imageUrl:
43
+ logger.info(f"Backfilling missing image for existing article: {article.title[:40]}")
44
+ doc_ref.update({
45
+ "imageUrl": article.imageUrl,
46
+ "hasImage": True
47
+ })
48
+ except Exception as e:
49
+ logger.error(f"Failed to save article {article.title}: {e}")
50
+
51
+ logger.info(f"Saved {len(new_articles)} new articles to Firestore.")
52
+ return new_articles
53
+
54
+ def exists(self, article_id: str) -> bool:
55
+ if not self.db:
56
+ return False
57
+ doc_ref = self.db.collection('articles').document(article_id)
58
+ return doc_ref.get().exists
domain/__pycache__/interfaces.cpython-314.pyc ADDED
Binary file (1.21 kB). View file
 
domain/__pycache__/models.cpython-314.pyc ADDED
Binary file (2.04 kB). View file
 
domain/entities/__pycache__/article.cpython-314.pyc ADDED
Binary file (5.69 kB). View file
 
domain/entities/article.py ADDED
@@ -0,0 +1,74 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from pydantic import BaseModel, HttpUrl, Field, field_validator
2
+ from typing import Optional, List
3
+ import datetime
4
+ import uuid
5
+ import hashlib
6
+
7
+ class Article(BaseModel):
8
+ id: str = Field(default_factory=lambda: str(uuid.uuid4()))
9
+ title: str
10
+ summary: str
11
+ url: str
12
+ imageUrl: Optional[str] = None
13
+ source: str
14
+ publishedAt: datetime.datetime
15
+ language: str = "ar"
16
+
17
+ # Enriched Fields
18
+ category: Optional[str] = None
19
+ subcategory: Optional[str] = None
20
+ region: Optional[str] = None
21
+ country: Optional[str] = None
22
+ city: Optional[str] = None
23
+
24
+ tags: List[str] = Field(default_factory=list)
25
+ keywords: List[str] = Field(default_factory=list)
26
+ people: List[str] = Field(default_factory=list)
27
+ organizations: List[str] = Field(default_factory=list)
28
+ locations: List[str] = Field(default_factory=list)
29
+
30
+ sentiment: Optional[str] = None
31
+ importance: Optional[int] = None
32
+ readingTime: Optional[int] = None
33
+ hasImage: bool = False
34
+
35
+ @classmethod
36
+ def normalize_url(cls, url: str) -> str:
37
+ """Strips http/https differences, www prefix, and tracking parameters to ensure 100% identical URL hashing."""
38
+ from urllib.parse import urlparse, parse_qs, urlunparse, urlencode
39
+ # Force https and strip www.
40
+ url_clean_scheme = url.replace('http://', 'https://')
41
+ parsed = urlparse(url_clean_scheme)
42
+ netloc = parsed.netloc.lower()
43
+ if netloc.startswith('www.'):
44
+ netloc = netloc[4:]
45
+
46
+ # Strip tracking query params
47
+ query_params = parse_qs(parsed.query, keep_blank_values=True)
48
+ filtered_params = {k: v for k, v in query_params.items() if not k.startswith('utm_') and k not in ['ref', 'fbclid', 'rss', 'src', 'amp']}
49
+ clean_query = urlencode(filtered_params, doseq=True)
50
+ # Reconstruct URL without trailing slash and tracking params
51
+ clean_url = urlunparse((parsed.scheme, netloc, parsed.path.rstrip('/'), parsed.params, clean_query, ''))
52
+ return clean_url
53
+
54
+ @classmethod
55
+ def generate_id_from_url(cls, url: str) -> str:
56
+ """Generates a deterministic ID from the normalized URL to prevent duplicates."""
57
+ clean_url = cls.normalize_url(url)
58
+ return hashlib.md5(clean_url.encode('utf-8')).hexdigest()
59
+
60
+ @classmethod
61
+ def generate_title_hash(cls, title: str) -> str:
62
+ """Generates a normalized hash of the title to detect cross-source headline duplicates."""
63
+ import re
64
+ clean_title = re.sub(r'[^\w\s]', '', title.lower()).strip()
65
+ return hashlib.md5(clean_title.encode('utf-8')).hexdigest()
66
+
67
+ @field_validator('readingTime', 'importance', mode='before')
68
+ def convert_float_to_int(cls, v):
69
+ if v is not None:
70
+ try:
71
+ return max(1, round(float(v)))
72
+ except (ValueError, TypeError):
73
+ return v
74
+ return v
domain/interfaces.py ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ from typing import List, Protocol
2
+ from domain.entities.article import Article
3
+
4
+ class ScraperInterface(Protocol):
5
+ source_name: str
6
+ language: str
7
+
8
+ def scrape(self) -> List[Article]:
9
+ """Scrapes the target website and returns a list of Articles."""
10
+ ...
domain/repositories/__pycache__/article_repository.cpython-314.pyc ADDED
Binary file (1.53 kB). View file
 
domain/repositories/article_repository.py ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from abc import ABC, abstractmethod
2
+ from typing import List
3
+ from domain.entities.article import Article
4
+
5
+ class ArticleRepository(ABC):
6
+ @abstractmethod
7
+ def save_articles(self, articles: List[Article]) -> List[Article]:
8
+ """Saves a list of articles and returns the list of newly saved articles."""
9
+ pass
10
+
11
+ @abstractmethod
12
+ def exists(self, article_id: str) -> bool:
13
+ """Checks if an article already exists by its ID."""
14
+ pass
domain/usecases/__pycache__/scrape_and_enrich.cpython-314.pyc ADDED
Binary file (4.29 kB). View file
 
domain/usecases/scrape_and_enrich.py ADDED
@@ -0,0 +1,69 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import logging
2
+ from typing import List
3
+ from domain.entities.article import Article
4
+ from domain.repositories.article_repository import ArticleRepository
5
+ from domain.interfaces import ScraperInterface
6
+ from services.ai_classifier import HuggingFaceClassifier
7
+ from services.duplicate_detector import DuplicateDetector
8
+
9
+ from services.lexicons import CATEGORY_IMAGES_MAP
10
+
11
+ logger = logging.getLogger(__name__)
12
+
13
+ class ScrapeAndEnrichUseCase:
14
+ def __init__(self,
15
+ repository: ArticleRepository,
16
+ scrapers: List[ScraperInterface],
17
+ classifier: HuggingFaceClassifier,
18
+ duplicate_detector: DuplicateDetector):
19
+ self.repository = repository
20
+ self.scrapers = scrapers
21
+ self.classifier = classifier
22
+ self.duplicate_detector = duplicate_detector
23
+
24
+ def execute(self) -> List[Article]:
25
+ all_articles = []
26
+
27
+ # 1. Scrape raw articles from all sources
28
+ for scraper in self.scrapers:
29
+ logger.info(f"Scraping raw feed from {scraper.source_name} ({scraper.language})...")
30
+ try:
31
+ articles = scraper.scrape()
32
+ logger.info(f"Found {len(articles)} raw articles from {scraper.source_name}")
33
+ all_articles.extend(articles)
34
+ except Exception as e:
35
+ logger.error(f"Failed to scrape {scraper.source_name}: {e}")
36
+
37
+ if not all_articles:
38
+ logger.info("No articles found across all scrapers.")
39
+ return []
40
+
41
+ # 2. Filter duplicates
42
+ new_articles = self.duplicate_detector.filter_new_articles(all_articles)
43
+ if not new_articles:
44
+ logger.info("No new articles to enrich.")
45
+ return []
46
+
47
+ # 3. Enrich articles via AI pipeline & apply Layer 3 Image Fallback
48
+ enriched_articles = []
49
+ logger.info(f"Starting AI enrichment for {len(new_articles)} new articles...")
50
+
51
+ for index, article in enumerate(new_articles):
52
+ logger.info(f"Enriching article {index + 1}/{len(new_articles)}: {article.title[:50]}...")
53
+ enriched = self.classifier.enrich_article(article)
54
+
55
+ # Layer 3: Ensure imageUrl is NEVER null
56
+ if not enriched.imageUrl:
57
+ fallback_img = CATEGORY_IMAGES_MAP.get(enriched.category, CATEGORY_IMAGES_MAP["General"])
58
+ enriched.imageUrl = fallback_img
59
+ enriched.hasImage = True
60
+ else:
61
+ enriched.hasImage = True
62
+
63
+ enriched_articles.append(enriched)
64
+
65
+ # 4. Save to Repository
66
+ logger.info(f"Saving {len(enriched_articles)} enriched articles to database...")
67
+ new_saved_articles = self.repository.save_articles(enriched_articles)
68
+
69
+ return new_saved_articles
main.py ADDED
@@ -0,0 +1,133 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import logging
2
+ from scrapers.rss_scraper import RssScraper
3
+ from data.repositories.article_repository_impl import FirestoreArticleRepository
4
+ from services.ai_classifier import HuggingFaceClassifier
5
+ from services.duplicate_detector import DuplicateDetector
6
+ from domain.usecases.scrape_and_enrich import ScrapeAndEnrichUseCase
7
+
8
+ logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(name)s - %(levelname)s - %(message)s')
9
+ logger = logging.getLogger(__name__)
10
+
11
+ # Configure our RSS sources across diverse categories (General, Tech, Gaming, Food, Sports, Business, Science, Entertainment)
12
+ SOURCES = [
13
+ # --- World & Middle East News (English) ---
14
+ RssScraper("BBC Middle East", "en", "http://feeds.bbci.co.uk/news/world/middle_east/rss.xml", category="Middle East"),
15
+ RssScraper("BBC World", "en", "http://feeds.bbci.co.uk/news/world/rss.xml", category="World"),
16
+ RssScraper("Middle East Eye", "en", "https://www.middleeasteye.net/rss", category="Middle East"),
17
+ RssScraper("Middle East Monitor", "en", "https://www.middleeastmonitor.com/feed/", category="Middle East"),
18
+ RssScraper("The Guardian Middle East", "en", "https://www.theguardian.com/world/middleeast/rss", category="Middle East"),
19
+ RssScraper("NY Times Middle East", "en", "https://rss.nytimes.com/services/xml/rss/nyt/MiddleEast.xml", category="Middle East"),
20
+ RssScraper("Al Jazeera English", "en", "https://www.aljazeera.com/xml/rss/all.xml", category="World"),
21
+
22
+ # --- Egyptian & Arab World News (Arabic) ---
23
+ RssScraper("Youm7", "ar", "http://www.youm7.com/rss/SectionRss?SectionID=65", category="Egypt"),
24
+ RssScraper("Al-Ahram", "ar", "https://gate.ahram.org.eg/NewsRss.aspx", category="Egypt"),
25
+ RssScraper("Al Masry Al Youm", "ar", "https://www.almasryalyoum.com/rss/rss", category="Egypt"),
26
+ RssScraper("Sada El Balad", "ar", "https://www.elbalad.news/rss.aspx", category="Egypt"),
27
+ RssScraper("Al Shorouk", "ar", "https://www.shorouknews.com/rss/news.aspx", category="Egypt"),
28
+ RssScraper("Al-Dustour", "ar", "https://www.dostor.org/rss.aspx", category="Egypt"),
29
+ RssScraper("Sky News Arabia", "ar", "https://www.skynewsarabia.com/rss", category="Middle East"),
30
+ RssScraper("RT Arabic", "ar", "https://arabic.rt.com/rss/", category="Middle East"),
31
+ RssScraper("Asharq Al-Awsat", "ar", "https://aawsat.com/feed", category="Middle East"),
32
+ RssScraper("Alhurra", "ar", "https://www.alhurra.com/rss", category="Middle East"),
33
+ RssScraper("BBC Arabic", "ar", "http://feeds.bbci.co.uk/arabic/rss.xml", category="Middle East"),
34
+ RssScraper("CNN Arabic", "ar", "https://arabic.cnn.com/api/v1/rss/middle_east/rss.xml", category="Middle East"),
35
+ RssScraper("France 24 Arabic", "ar", "https://www.france24.com/ar/rss", category="Middle East"),
36
+ RssScraper("Hespress", "ar", "https://www.hespress.com/feed", category="North Africa"),
37
+ RssScraper("Echorouk", "ar", "https://www.echoroukonline.com/feed", category="North Africa"),
38
+
39
+ # --- Technology & AI ---
40
+ RssScraper("TechCrunch", "en", "https://techcrunch.com/feed/", category="Technology"),
41
+ RssScraper("Wired", "en", "https://www.wired.com/feed/rss", category="Technology"),
42
+ RssScraper("The Verge", "en", "https://www.theverge.com/rss/index.xml", category="Technology"),
43
+ RssScraper("Ars Technica", "en", "https://feeds.arstechnica.com/arstechnica/index", category="Technology"),
44
+ RssScraper("Engadget", "en", "https://www.engadget.com/rss.xml", category="Technology"),
45
+ RssScraper("Aitnews", "ar", "https://aitnews.com/feed/", category="Technology"),
46
+ RssScraper("Unlimit Tech", "ar", "https://unlimit-tech.com/feed/", category="Technology"),
47
+
48
+ # --- Gaming & Esports ---
49
+ RssScraper("IGN", "en", "https://feeds.feedburner.com/ign/all", category="Gaming"),
50
+ RssScraper("Eurogamer", "en", "https://www.eurogamer.net/?format=rss", category="Gaming"),
51
+ RssScraper("GameSpot", "en", "https://www.gamespot.com/feeds/news/", category="Gaming"),
52
+ RssScraper("Polygon", "en", "https://www.polygon.com/rss/index.xml", category="Gaming"),
53
+ RssScraper("Saudi Gamer", "ar", "https://saudigamer.com/feed/", category="Gaming"),
54
+ RssScraper("TrueGaming", "ar", "https://www.true-gaming.net/home/feed/", category="Gaming"),
55
+
56
+ # --- Food, Cooking & Lifestyle ---
57
+ RssScraper("Serious Eats", "en", "https://www.seriouseats.com/feed", category="Food"),
58
+ RssScraper("BBC Good Food", "en", "https://www.bbcgoodfood.com/feed/rss2", category="Food"),
59
+ RssScraper("Food52", "en", "https://food52.com/blog.rss", category="Food"),
60
+ RssScraper("SuperMama", "ar", "https://www.supermama.me/rss.xml", category="Lifestyle"),
61
+
62
+ # --- Business, Economy & Finance ---
63
+ RssScraper("CNBC World", "en", "https://www.cnbc.com/id/100727362/device/rss/rss.html", category="Business"),
64
+ RssScraper("Financial Times", "en", "https://www.ft.com/?format=rss", category="Business"),
65
+ RssScraper("RT Arabic Business", "ar", "https://arabic.rt.com/rss/business/", category="Business"),
66
+ RssScraper("Al Mal News", "ar", "https://almalnews.com/feed/", category="Business"),
67
+
68
+ # --- Sports ---
69
+ RssScraper("BBC Sport", "en", "http://feeds.bbci.co.uk/sport/rss.xml", category="Sports"),
70
+ RssScraper("RT Arabic Sports", "ar", "https://arabic.rt.com/rss/sport/", category="Sports"),
71
+ RssScraper("FilGoal", "ar", "https://www.filgoal.com/rss/news", category="Sports"),
72
+
73
+ # --- Science & Space ---
74
+ RssScraper("ScienceDaily", "en", "https://www.sciencedaily.com/rss/all.xml", category="Science"),
75
+ RssScraper("NASA News", "en", "https://www.nasa.gov/news-release/feed/", category="Science"),
76
+ RssScraper("RT Arabic Science", "ar", "https://arabic.rt.com/rss/space/", category="Science"),
77
+
78
+ # --- Entertainment & Culture ---
79
+ RssScraper("Variety", "en", "https://variety.com/feed/", category="Entertainment"),
80
+ RssScraper("Hollywood Reporter", "en", "https://www.hollywoodreporter.com/feed/", category="Entertainment")
81
+ ]
82
+
83
+ def job():
84
+ logger.info("Starting scraping job with AI Enrichment...")
85
+
86
+ # 1. Initialize dependencies
87
+ try:
88
+ repo = FirestoreArticleRepository(service_account_path="service-account.json")
89
+ except Exception as e:
90
+ logger.error(f"Failed to initialize Firestore: {e}")
91
+ return
92
+
93
+ if not repo.db:
94
+ logger.error("Skipping job because Firestore is not initialized. Please ensure service-account.json exists.")
95
+ return
96
+
97
+ classifier = HuggingFaceClassifier()
98
+ duplicate_detector = DuplicateDetector(repository=repo)
99
+
100
+ # 2. Setup Usecase
101
+ usecase = ScrapeAndEnrichUseCase(
102
+ repository=repo,
103
+ scrapers=SOURCES,
104
+ classifier=classifier,
105
+ duplicate_detector=duplicate_detector
106
+ )
107
+
108
+ # 3. Execute Pipeline
109
+ new_articles = usecase.execute()
110
+
111
+ # 4. Send Notifications
112
+ if new_articles:
113
+ try:
114
+ from firebase_admin import messaging
115
+ top_article = new_articles[0]
116
+ message = messaging.Message(
117
+ notification=messaging.Notification(
118
+ title=f"New story from {top_article.source}",
119
+ body=top_article.title
120
+ ),
121
+ topic='daily_news'
122
+ )
123
+ response = messaging.send(message)
124
+ logger.info(f"Successfully sent FCM notification: {response}")
125
+ except Exception as e:
126
+ logger.error(f"Failed to send FCM notification: {e}")
127
+ else:
128
+ logger.info("No new articles processed.")
129
+
130
+ if __name__ == "__main__":
131
+ logger.info("News Intelligence Engine started.")
132
+ job()
133
+ logger.info("Scraping job finished.")
nlp_training/__init__.py ADDED
@@ -0,0 +1 @@
 
 
1
+ # NLP Training Scripts for WhatsNew News Intelligence Engine
nlp_training/prepare_datasets.py ADDED
@@ -0,0 +1,398 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Dataset Preparation Utility for WhatsNew NLP Training.
3
+
4
+ Downloads and prepares Arabic + English datasets for fine-tuning:
5
+ - SANAD + HuffPost → Category Classification
6
+ - ASAD + SST-2 → Sentiment Analysis
7
+ - WikiANN → Named Entity Recognition
8
+
9
+ Usage:
10
+ python nlp_training/prepare_datasets.py --task all
11
+ python nlp_training/prepare_datasets.py --task category
12
+ python nlp_training/prepare_datasets.py --task sentiment
13
+ python nlp_training/prepare_datasets.py --task ner
14
+ """
15
+
16
+ import os
17
+ import argparse
18
+ import json
19
+ import logging
20
+
21
+ logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
22
+ logger = logging.getLogger(__name__)
23
+
24
+ # Our unified 13 categories (must match CATEGORIES_MAP keys)
25
+ UNIFIED_CATEGORIES = [
26
+ "Sports", "Technology", "Gaming", "Business", "Health", "Science",
27
+ "Food", "Entertainment", "Politics", "Education", "Crime", "Weather", "Travel"
28
+ ]
29
+
30
+ # -----------------------------------------------------------------------
31
+ # Label mapping: Map source dataset labels → our unified category labels
32
+ # -----------------------------------------------------------------------
33
+
34
+ # SANAD Arabic dataset label → our category
35
+ SANAD_LABEL_MAP = {
36
+ "culture": "Entertainment",
37
+ "finance": "Business",
38
+ "medical": "Health",
39
+ "politics": "Politics",
40
+ "religion": "Education",
41
+ "sports": "Sports",
42
+ "tech": "Technology",
43
+ }
44
+
45
+ # HuffPost English dataset label → our category
46
+ HUFFPOST_LABEL_MAP = {
47
+ "POLITICS": "Politics",
48
+ "WELLNESS": "Health",
49
+ "ENTERTAINMENT": "Entertainment",
50
+ "TRAVEL": "Travel",
51
+ "STYLE & BEAUTY": "Entertainment",
52
+ "PARENTING": "Education",
53
+ "HEALTHY LIVING": "Health",
54
+ "QUEER VOICES": "Entertainment",
55
+ "FOOD & DRINK": "Food",
56
+ "BUSINESS": "Business",
57
+ "COMEDY": "Entertainment",
58
+ "SPORTS": "Sports",
59
+ "BLACK VOICES": "Politics",
60
+ "HOME & LIVING": "Food",
61
+ "PARENTS": "Education",
62
+ "THE WORLDPOST": "Politics",
63
+ "WEDDINGS": "Entertainment",
64
+ "WOMEN": "Politics",
65
+ "IMPACT": "Politics",
66
+ "DIVORCE": "Entertainment",
67
+ "CRIME": "Crime",
68
+ "MEDIA": "Entertainment",
69
+ "WEIRD NEWS": "Entertainment",
70
+ "GREEN": "Science",
71
+ "WORLDPOST": "Politics",
72
+ "RELIGION": "Education",
73
+ "SCIENCE": "Science",
74
+ "TECH": "Technology",
75
+ "TASTE": "Food",
76
+ "MONEY": "Business",
77
+ "ARTS": "Entertainment",
78
+ "FIFTY": "Entertainment",
79
+ "GOOD NEWS": "Entertainment",
80
+ "ARTS & CULTURE": "Entertainment",
81
+ "ENVIRONMENT": "Science",
82
+ "COLLEGE": "Education",
83
+ "LATINO VOICES": "Politics",
84
+ "CULTURE & ARTS": "Entertainment",
85
+ "EDUCATION": "Education",
86
+ "STYLE": "Entertainment",
87
+ }
88
+
89
+ SENTIMENT_LABELS = ["Positive", "Negative", "Neutral"]
90
+
91
+
92
+ def prepare_category_datasets(output_dir: str):
93
+ """
94
+ Download and prepare news category classification datasets (Arabic + English).
95
+ Outputs: category_train.jsonl, category_test.jsonl with {text, label, language} pairs.
96
+ """
97
+ from datasets import load_dataset
98
+
99
+ os.makedirs(output_dir, exist_ok=True)
100
+ raw_dir = "nlp_training/raw_data"
101
+ os.makedirs(raw_dir, exist_ok=True)
102
+ all_samples = []
103
+
104
+ # --- Local Kaggle Raw Data Check ---
105
+ for filename in os.listdir(raw_dir):
106
+ filepath = os.path.join(raw_dir, filename)
107
+ if filename.endswith(".csv"):
108
+ try:
109
+ import pandas as pd
110
+ df = pd.read_csv(filepath)
111
+ text_col = next((c for c in df.columns if c.lower() in ["text", "article", "content", "headline"]), None)
112
+ label_col = next((c for c in df.columns if c.lower() in ["label", "category", "class", "topic"]), None)
113
+ if text_col and label_col:
114
+ loaded = 0
115
+ for _, row in df.iterrows():
116
+ text = str(row[text_col])
117
+ raw_label = str(row[label_col]).lower()
118
+ mapped = SANAD_LABEL_MAP.get(raw_label) or HUFFPOST_LABEL_MAP.get(raw_label.upper())
119
+ if mapped and text:
120
+ lang = "ar" if any('\u0600' <= char <= '\u06FF' for char in text[:100]) else "en"
121
+ all_samples.append({"text": text[:512], "label": mapped, "language": lang})
122
+ loaded += 1
123
+ logger.info(f"Loaded {loaded} samples from local CSV '{filename}'")
124
+ except Exception as e:
125
+ logger.warning(f"Failed parsing local file {filename}: {e}")
126
+
127
+ # --- English: AG News ---
128
+ logger.info("Loading AG News English dataset...")
129
+ ag_news_map = {0: "Politics", 1: "Sports", 2: "Business", 3: "Technology"}
130
+ try:
131
+ ag_ds = load_dataset("ag_news", split="train")
132
+ en_count = 0
133
+ for sample in ag_ds:
134
+ text = sample.get("text", "")
135
+ label_id = sample.get("label")
136
+ mapped = ag_news_map.get(label_id)
137
+ if mapped and text:
138
+ all_samples.append({"text": text[:512], "label": mapped, "language": "en"})
139
+ en_count += 1
140
+ logger.info(f"Loaded {en_count} English samples from AG News")
141
+ except Exception as e:
142
+ logger.warning(f"Could not load AG News: {e}")
143
+
144
+ # --- English: 20 Newsgroups ---
145
+ try:
146
+ hp_ds = load_dataset("SetFit/20_newsgroups", split="train")
147
+ newsgroups_map = {
148
+ "rec.sport.baseball": "Sports", "rec.sport.hockey": "Sports",
149
+ "comp.graphics": "Technology", "comp.sys.mac.hardware": "Technology",
150
+ "talk.politics.mideast": "Politics", "talk.politics.misc": "Politics",
151
+ "sci.space": "Science", "sci.med": "Health"
152
+ }
153
+ for sample in hp_ds:
154
+ text = sample.get("text", "")
155
+ label_text = sample.get("label_text", "")
156
+ mapped = newsgroups_map.get(label_text)
157
+ if mapped and text:
158
+ all_samples.append({"text": text[:512], "label": mapped, "language": "en"})
159
+ except Exception:
160
+ pass
161
+
162
+ # --- Arabic: Arabic News Datasets ---
163
+ logger.info("Loading Arabic news classification dataset...")
164
+ arabic_loaded = False
165
+ arabic_ds_candidates = ["arbml/Arabic_News_Tweets", "he3als/sanad", "ahedag/arabic_news"]
166
+
167
+ for candidate in arabic_ds_candidates:
168
+ try:
169
+ ar_ds = load_dataset(candidate, split="train")
170
+ ar_count = 0
171
+ for sample in ar_ds:
172
+ text = sample.get("text") or sample.get("article") or sample.get("tweet") or ""
173
+ raw_label = str(sample.get("label") or sample.get("category") or "").lower()
174
+ mapped = SANAD_LABEL_MAP.get(raw_label)
175
+ if not mapped:
176
+ if any(k in raw_label for k in ["رياض", "sport"]): mapped = "Sports"
177
+ elif any(k in raw_label for k in ["تقن", "tech"]): mapped = "Technology"
178
+ elif any(k in raw_label for k in ["سياس", "politic"]): mapped = "Politics"
179
+ elif any(k in raw_label for k in ["اقتصاد", "مال", "econ"]): mapped = "Business"
180
+ elif any(k in raw_label for k in ["صح", "health"]): mapped = "Health"
181
+ elif any(k in raw_label for k in ["فن", "ثقاف", "entertain"]): mapped = "Entertainment"
182
+
183
+ if mapped and text:
184
+ all_samples.append({"text": text[:512], "label": mapped, "language": "ar"})
185
+ ar_count += 1
186
+ if ar_count > 0:
187
+ logger.info(f"Loaded {ar_count} Arabic news samples from '{candidate}'")
188
+ arabic_loaded = True
189
+ break
190
+ except Exception:
191
+ continue
192
+
193
+ if not arabic_loaded and len([s for s in all_samples if s["language"] == "ar"]) < 10:
194
+ logger.info("Using built-in Arabic news samples generator for fallback testing...")
195
+ sample_ar_news = [
196
+ ("الأهلي يفوز على الزمالك في قمة الدوري الممتاز", "Sports"),
197
+ ("إطلاق جديد لهاتف آيفون يدعم تقنيات الذكاء الاصطناعي", "Technology"),
198
+ ("ارتفاع أسعار النفط والأسهم في البورصة العالمية", "Business"),
199
+ ("علماء يكتشفون علاجاً جديداً يرفع المناعة ضد الفيروسات", "Health"),
200
+ ("مفاوضات سياسية جديدة في قمة مجلس الأمن الدولي", "Politics"),
201
+ ("سوني تعلن عن ألعاب جديدة لجهاز بلايستيشن", "Gaming"),
202
+ ("وكالة ناسا تطلق تلسكوب فضاء جديد لاستكشاف المريخ", "Science"),
203
+ ("عرض أول لفيلم سينمائي جديد يتصدر شباك التذاكر", "Entertainment"),
204
+ ]
205
+ for text, cat in sample_ar_news * 50:
206
+ all_samples.append({"text": text, "label": cat, "language": "ar"})
207
+
208
+ # --- Shuffle and split 90/10 ---
209
+ import random
210
+ random.shuffle(all_samples)
211
+ split_idx = int(len(all_samples) * 0.9)
212
+ train_data = all_samples[:split_idx]
213
+ test_data = all_samples[split_idx:]
214
+
215
+ _save_jsonl(train_data, os.path.join(output_dir, "category_train.jsonl"))
216
+ _save_jsonl(test_data, os.path.join(output_dir, "category_test.jsonl"))
217
+ logger.info(f"Category dataset saved: {len(train_data)} train, {len(test_data)} test")
218
+
219
+
220
+ def prepare_sentiment_datasets(output_dir: str):
221
+ """
222
+ Download and prepare Sentiment Analysis datasets (Arabic + English).
223
+ Outputs: sentiment_train.jsonl, sentiment_test.jsonl with {text, label, language} pairs.
224
+ """
225
+ from datasets import load_dataset
226
+
227
+ os.makedirs(output_dir, exist_ok=True)
228
+ all_samples = []
229
+
230
+ # --- English: SST-2 / TweetEval ---
231
+ logger.info("Loading English sentiment dataset...")
232
+ en_loaded = False
233
+ for candidate in [("stanfordnlp/sst2", None), ("nyu-mll/glue", "sst2"), ("mteb/tweet_eval_sentiment", None)]:
234
+ try:
235
+ ds_name, sub = candidate
236
+ sst2 = load_dataset(ds_name, sub, split="train") if sub else load_dataset(ds_name, split="train")
237
+ en_count = 0
238
+ for sample in sst2:
239
+ text = sample.get("sentence") or sample.get("text") or ""
240
+ label = sample.get("label", 0)
241
+ mapped = "Positive" if label == 1 else "Negative" if label == 0 else "Neutral"
242
+ if text:
243
+ all_samples.append({"text": text[:512], "label": mapped, "language": "en"})
244
+ en_count += 1
245
+ if en_count > 0:
246
+ logger.info(f"Loaded {en_count} English sentiment samples from '{ds_name}'")
247
+ en_loaded = True
248
+ break
249
+ except Exception:
250
+ continue
251
+
252
+ # --- Arabic: TweetEval / Sentiment ---
253
+ logger.info("Loading Arabic sentiment dataset...")
254
+ ar_loaded = False
255
+ for candidate in ["mteb/tweet_eval_sentiment", "MohamedAtef/LABR"]:
256
+ try:
257
+ asad = load_dataset(candidate, split="train")
258
+ ar_count = 0
259
+ for sample in asad:
260
+ text = sample.get("text") or sample.get("text_raw") or ""
261
+ label = sample.get("label", 1)
262
+ label_map = {0: "Negative", 1: "Neutral", 2: "Positive"}
263
+ mapped = label_map.get(label, "Neutral")
264
+ if text:
265
+ all_samples.append({"text": text[:512], "label": mapped, "language": "ar"})
266
+ ar_count += 1
267
+ if ar_count > 0:
268
+ logger.info(f"Loaded {ar_count} Arabic sentiment samples")
269
+ ar_loaded = True
270
+ break
271
+ except Exception:
272
+ continue
273
+
274
+ if not ar_loaded and len([s for s in all_samples if s["language"] == "ar"]) < 10:
275
+ sample_ar_sentiment = [
276
+ ("فوز تاريخي وانتعاش اقتصادي ممتاز للمنطقة", "Positive"),
277
+ ("تراجع كبير وخسائر ضخمة وتدهور الأوضاع", "Negative"),
278
+ ("عقد اجتماع عادي لمناقشة جدول الأعمال", "Neutral"),
279
+ ]
280
+ for text, sent in sample_ar_sentiment * 100:
281
+ all_samples.append({"text": text, "label": sent, "language": "ar"})
282
+
283
+ # --- Shuffle and split ---
284
+ import random
285
+ random.shuffle(all_samples)
286
+ split_idx = int(len(all_samples) * 0.9)
287
+
288
+ _save_jsonl(all_samples[:split_idx], os.path.join(output_dir, "sentiment_train.jsonl"))
289
+ _save_jsonl(all_samples[split_idx:], os.path.join(output_dir, "sentiment_test.jsonl"))
290
+ logger.info(f"Sentiment dataset saved: {split_idx} train, {len(all_samples) - split_idx} test")
291
+
292
+
293
+ def prepare_ner_datasets(output_dir: str):
294
+ """
295
+ Download and prepare WikiANN / CoNLL-2003 for NER.
296
+ Outputs: ner_train.jsonl, ner_test.jsonl with {tokens, ner_tags, language} pairs.
297
+ """
298
+ from datasets import load_dataset
299
+
300
+ os.makedirs(output_dir, exist_ok=True)
301
+ all_train = []
302
+ all_test = []
303
+
304
+ # --- WikiANN Arabic ---
305
+ logger.info("Loading WikiANN Arabic NER dataset...")
306
+ try:
307
+ wikiann_ar = load_dataset("unimelb-nlp/wikiann", "ar")
308
+ ar_train = [{"tokens": s["tokens"], "ner_tags": s["ner_tags"], "language": "ar"} for s in wikiann_ar["train"]]
309
+ ar_test = [{"tokens": s["tokens"], "ner_tags": s["ner_tags"], "language": "ar"} for s in wikiann_ar["test"]]
310
+ all_train.extend(ar_train)
311
+ all_test.extend(ar_test)
312
+ logger.info(f"Loaded WikiANN Arabic: {len(ar_train)} train, {len(ar_test)} test")
313
+ except Exception as e:
314
+ logger.warning(f"Could not load WikiANN Arabic: {e}")
315
+
316
+ # --- WikiANN English ---
317
+ logger.info("Loading WikiANN English NER dataset...")
318
+ try:
319
+ wikiann_en = load_dataset("unimelb-nlp/wikiann", "en")
320
+ en_train = [{"tokens": s["tokens"], "ner_tags": s["ner_tags"], "language": "en"} for s in wikiann_en["train"]]
321
+ en_test = [{"tokens": s["tokens"], "ner_tags": s["ner_tags"], "language": "en"} for s in wikiann_en["test"]]
322
+ all_train.extend(en_train)
323
+ all_test.extend(en_test)
324
+ logger.info(f"Loaded WikiANN English: {len(en_train)} train, {len(en_test)} test")
325
+ except Exception as e:
326
+ logger.warning(f"Could not load WikiANN English: {e}")
327
+
328
+ # --- CoNLL-2003 English ---
329
+ logger.info("Loading CoNLL-2003 English NER dataset...")
330
+ try:
331
+ conll = load_dataset("eriktks/conll2003")
332
+ conll_train = [{"tokens": s["tokens"], "ner_tags": s["ner_tags"], "language": "en"} for s in conll["train"]]
333
+ conll_test = [{"tokens": s["tokens"], "ner_tags": s["ner_tags"], "language": "en"} for s in conll["test"]]
334
+ all_train.extend(conll_train)
335
+ all_test.extend(conll_test)
336
+ logger.info(f"Added CoNLL-2003: {len(conll_train)} train, {len(conll_test)} test")
337
+ except Exception as e:
338
+ logger.warning(f"Could not load CoNLL-2003: {e}")
339
+
340
+ # Fallback if datasets were unavailable
341
+ if not all_train:
342
+ sample_ner = [
343
+ {"tokens": ["الرئيس", "السيسي", "في", "القاهرة"], "ner_tags": [0, 1, 0, 5], "language": "ar"},
344
+ {"tokens": ["Biden", "visited", "London"], "ner_tags": [1, 0, 5], "language": "en"}
345
+ ]
346
+ all_train = sample_ner * 50
347
+ all_test = sample_ner * 10
348
+
349
+ _save_jsonl(all_train, os.path.join(output_dir, "ner_train.jsonl"))
350
+ _save_jsonl(all_test, os.path.join(output_dir, "ner_test.jsonl"))
351
+ logger.info(f"NER dataset saved: {len(all_train)} train, {len(all_test)} test")
352
+
353
+
354
+ def _save_jsonl(data: list, filepath: str):
355
+ """Save a list of dicts as JSONL, merging with any existing file to preserve collected data."""
356
+ existing_samples = []
357
+ if os.path.exists(filepath):
358
+ try:
359
+ with open(filepath, 'r', encoding='utf-8') as f:
360
+ for line in f:
361
+ if line.strip():
362
+ existing_samples.append(json.loads(line.strip()))
363
+ except Exception:
364
+ pass
365
+
366
+ # Merge and deduplicate by text/tokens content
367
+ seen_keys = set()
368
+ combined_data = []
369
+
370
+ for item in existing_samples + data:
371
+ key = str(item.get("text") or item.get("tokens") or "")
372
+ if key and key not in seen_keys:
373
+ seen_keys.add(key)
374
+ combined_data.append(item)
375
+
376
+ with open(filepath, 'w', encoding='utf-8') as f:
377
+ for item in combined_data:
378
+ f.write(json.dumps(item, ensure_ascii=False) + '\n')
379
+
380
+ new_added = len(combined_data) - len(existing_samples)
381
+ logger.info(f"Saved {len(combined_data)} total samples ({len(existing_samples)} existing + {max(0, new_added)} new) to {filepath}")
382
+
383
+
384
+ if __name__ == "__main__":
385
+ parser = argparse.ArgumentParser(description="Prepare NLP training datasets for WhatsNew")
386
+ parser.add_argument("--task", choices=["category", "sentiment", "ner", "all"], default="all",
387
+ help="Which dataset to prepare")
388
+ parser.add_argument("--output-dir", default="nlp_training/data", help="Output directory for processed datasets")
389
+ args = parser.parse_args()
390
+
391
+ if args.task in ("category", "all"):
392
+ prepare_category_datasets(args.output_dir)
393
+ if args.task in ("sentiment", "all"):
394
+ prepare_sentiment_datasets(args.output_dir)
395
+ if args.task in ("ner", "all"):
396
+ prepare_ner_datasets(args.output_dir)
397
+
398
+ logger.info("Dataset preparation complete!")
nlp_training/train_category_classifier.py ADDED
@@ -0,0 +1,190 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Fine-tune a multilingual category classifier for WhatsNew News Intelligence Engine.
3
+
4
+ Model: XLM-RoBERTa-base (handles Arabic + English in a single model)
5
+ Dataset: SANAD (Arabic) + HuffPost (English) → 13 unified news categories
6
+
7
+ Usage:
8
+ # 1. First prepare the data:
9
+ python nlp_training/prepare_datasets.py --task category
10
+
11
+ # 2. Then fine-tune:
12
+ python nlp_training/train_category_classifier.py
13
+
14
+ # 3. The fine-tuned model will be saved to ./models/category_classifier/
15
+ # and can be uploaded to HuggingFace Hub with:
16
+ python nlp_training/train_category_classifier.py --push-to-hub AbdoCrow/whatsnew-category
17
+
18
+ Requirements:
19
+ pip install transformers datasets torch scikit-learn accelerate
20
+ """
21
+
22
+ import os
23
+ import json
24
+ import logging
25
+ import argparse
26
+ import numpy as np
27
+
28
+ logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
29
+ logger = logging.getLogger(__name__)
30
+
31
+ # Our unified 13 categories
32
+ LABEL_LIST = [
33
+ "Sports", "Technology", "Gaming", "Business", "Health", "Science",
34
+ "Food", "Entertainment", "Politics", "Education", "Crime", "Weather", "Travel"
35
+ ]
36
+ LABEL2ID = {label: i for i, label in enumerate(LABEL_LIST)}
37
+ ID2LABEL = {i: label for i, label in enumerate(LABEL_LIST)}
38
+
39
+
40
+ def load_jsonl(filepath):
41
+ """Load a JSONL dataset file."""
42
+ samples = []
43
+ with open(filepath, 'r', encoding='utf-8') as f:
44
+ for line in f:
45
+ samples.append(json.loads(line.strip()))
46
+ return samples
47
+
48
+
49
+ def main(args):
50
+ from transformers import (
51
+ AutoTokenizer,
52
+ AutoModelForSequenceClassification,
53
+ TrainingArguments,
54
+ Trainer,
55
+ EarlyStoppingCallback
56
+ )
57
+ from datasets import Dataset
58
+ from sklearn.metrics import accuracy_score, f1_score
59
+
60
+ # ---------------------------------------------------------------
61
+ # 1. Load prepared datasets
62
+ # ---------------------------------------------------------------
63
+ data_dir = args.data_dir
64
+ train_path = os.path.join(data_dir, "category_train.jsonl")
65
+ test_path = os.path.join(data_dir, "category_test.jsonl")
66
+
67
+ if not os.path.exists(train_path):
68
+ logger.error(f"Training data not found at {train_path}. Run prepare_datasets.py first.")
69
+ return
70
+
71
+ train_data = load_jsonl(train_path)
72
+ test_data = load_jsonl(test_path)
73
+
74
+ # Filter to only include our known labels
75
+ train_data = [s for s in train_data if s["label"] in LABEL2ID]
76
+ test_data = [s for s in test_data if s["label"] in LABEL2ID]
77
+
78
+ logger.info(f"Train: {len(train_data)} samples, Test: {len(test_data)} samples")
79
+
80
+ # Convert to HuggingFace Dataset
81
+ train_dataset = Dataset.from_list([
82
+ {"text": s["text"], "label": LABEL2ID[s["label"]]} for s in train_data
83
+ ])
84
+ test_dataset = Dataset.from_list([
85
+ {"text": s["text"], "label": LABEL2ID[s["label"]]} for s in test_data
86
+ ])
87
+
88
+ # ---------------------------------------------------------------
89
+ # 2. Tokenize
90
+ # ---------------------------------------------------------------
91
+ model_name = args.model_name
92
+ tokenizer = AutoTokenizer.from_pretrained(model_name)
93
+
94
+ def tokenize_fn(examples):
95
+ return tokenizer(examples["text"], truncation=True, padding="max_length", max_length=256)
96
+
97
+ train_dataset = train_dataset.map(tokenize_fn, batched=True, remove_columns=["text"])
98
+ test_dataset = test_dataset.map(tokenize_fn, batched=True, remove_columns=["text"])
99
+
100
+ train_dataset.set_format("torch")
101
+ test_dataset.set_format("torch")
102
+
103
+ # ---------------------------------------------------------------
104
+ # 3. Load Model
105
+ # ---------------------------------------------------------------
106
+ model = AutoModelForSequenceClassification.from_pretrained(
107
+ model_name,
108
+ num_labels=len(LABEL_LIST),
109
+ id2label=ID2LABEL,
110
+ label2id=LABEL2ID
111
+ )
112
+ logger.info(f"Loaded model: {model_name} with {len(LABEL_LIST)} labels")
113
+
114
+ # ---------------------------------------------------------------
115
+ # 4. Training Configuration
116
+ # ---------------------------------------------------------------
117
+ output_dir = args.output_dir
118
+
119
+ training_args = TrainingArguments(
120
+ output_dir=output_dir,
121
+ num_train_epochs=args.epochs,
122
+ per_device_train_batch_size=args.batch_size,
123
+ per_device_eval_batch_size=args.batch_size * 2,
124
+ learning_rate=args.learning_rate,
125
+ weight_decay=0.01,
126
+ eval_strategy="epoch",
127
+ save_strategy="epoch",
128
+ load_best_model_at_end=True,
129
+ metric_for_best_model="f1_macro",
130
+ greater_is_better=True,
131
+ save_total_limit=2,
132
+ logging_steps=100,
133
+ warmup_ratio=0.1,
134
+ fp16=args.fp16,
135
+ report_to="none", # Disable wandb/tensorboard
136
+ )
137
+
138
+ def compute_metrics(eval_pred):
139
+ logits, labels = eval_pred
140
+ predictions = np.argmax(logits, axis=-1)
141
+ acc = accuracy_score(labels, predictions)
142
+ f1 = f1_score(labels, predictions, average="macro")
143
+ return {"accuracy": acc, "f1_macro": f1}
144
+
145
+ trainer = Trainer(
146
+ model=model,
147
+ args=training_args,
148
+ train_dataset=train_dataset,
149
+ eval_dataset=test_dataset,
150
+ compute_metrics=compute_metrics,
151
+ callbacks=[EarlyStoppingCallback(early_stopping_patience=2)]
152
+ )
153
+
154
+ # ---------------------------------------------------------------
155
+ # 5. Train
156
+ # ---------------------------------------------------------------
157
+ logger.info("Starting training...")
158
+ trainer.train()
159
+
160
+ # ---------------------------------------------------------------
161
+ # 6. Evaluate
162
+ # ---------------------------------------------------------------
163
+ results = trainer.evaluate()
164
+ logger.info(f"Evaluation results: {results}")
165
+
166
+ # ---------------------------------------------------------------
167
+ # 7. Save
168
+ # ---------------------------------------------------------------
169
+ trainer.save_model(output_dir)
170
+ tokenizer.save_pretrained(output_dir)
171
+ logger.info(f"Model saved to {output_dir}")
172
+
173
+ # Optionally push to HuggingFace Hub
174
+ if args.push_to_hub:
175
+ trainer.push_to_hub(args.push_to_hub)
176
+ logger.info(f"Model pushed to HuggingFace Hub: {args.push_to_hub}")
177
+
178
+
179
+ if __name__ == "__main__":
180
+ parser = argparse.ArgumentParser(description="Fine-tune a news category classifier")
181
+ parser.add_argument("--model-name", default="xlm-roberta-base", help="Base model name")
182
+ parser.add_argument("--data-dir", default="nlp_training/data", help="Directory with prepared datasets")
183
+ parser.add_argument("--output-dir", default="models/category_classifier", help="Output directory for trained model")
184
+ parser.add_argument("--epochs", type=int, default=5, help="Number of training epochs")
185
+ parser.add_argument("--batch-size", type=int, default=16, help="Training batch size")
186
+ parser.add_argument("--learning-rate", type=float, default=2e-5, help="Learning rate")
187
+ parser.add_argument("--fp16", action="store_true", help="Use mixed precision training (requires CUDA)")
188
+ parser.add_argument("--push-to-hub", type=str, default=None, help="HuggingFace Hub repo to push to")
189
+ args = parser.parse_args()
190
+ main(args)
nlp_training/train_ner_model.py ADDED
@@ -0,0 +1,271 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Fine-tune a multilingual NER model for WhatsNew News Intelligence Engine.
3
+
4
+ Model: XLM-RoBERTa-base
5
+ Dataset: WikiANN (Arabic + English) + CoNLL-2003 (English)
6
+ Entities: PER (people), ORG (organizations), LOC (locations)
7
+
8
+ Usage:
9
+ # 1. First prepare the data:
10
+ python nlp_training/prepare_datasets.py --task ner
11
+
12
+ # 2. Then fine-tune:
13
+ python nlp_training/train_ner_model.py
14
+
15
+ # 3. The fine-tuned model will be saved to ./models/ner_model/
16
+
17
+ Requirements:
18
+ pip install transformers datasets torch scikit-learn seqeval accelerate
19
+ """
20
+
21
+ import os
22
+ import json
23
+ import logging
24
+ import argparse
25
+ import numpy as np
26
+
27
+ logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
28
+ logger = logging.getLogger(__name__)
29
+
30
+ # WikiANN NER tag mapping (IOB2 format)
31
+ # WikiANN uses: O=0, B-PER=1, I-PER=2, B-ORG=3, I-ORG=4, B-LOC=5, I-LOC=6
32
+ LABEL_LIST = ["O", "B-PER", "I-PER", "B-ORG", "I-ORG", "B-LOC", "I-LOC"]
33
+ LABEL2ID = {label: i for i, label in enumerate(LABEL_LIST)}
34
+ ID2LABEL = {i: label for i, label in enumerate(LABEL_LIST)}
35
+
36
+
37
+ def load_jsonl(filepath):
38
+ samples = []
39
+ with open(filepath, 'r', encoding='utf-8') as f:
40
+ for line in f:
41
+ samples.append(json.loads(line.strip()))
42
+ return samples
43
+
44
+
45
+ def main(args):
46
+ from transformers import (
47
+ AutoTokenizer,
48
+ AutoModelForTokenClassification,
49
+ TrainingArguments,
50
+ Trainer,
51
+ DataCollatorForTokenClassification,
52
+ EarlyStoppingCallback
53
+ )
54
+ from datasets import Dataset
55
+
56
+ # ---------------------------------------------------------------
57
+ # 1. Load prepared datasets
58
+ # ---------------------------------------------------------------
59
+ data_dir = args.data_dir
60
+ train_path = os.path.join(data_dir, "ner_train.jsonl")
61
+ test_path = os.path.join(data_dir, "ner_test.jsonl")
62
+
63
+ if not os.path.exists(train_path):
64
+ logger.error(f"Training data not found at {train_path}. Run prepare_datasets.py first.")
65
+ return
66
+
67
+ train_data = load_jsonl(train_path)
68
+ test_data = load_jsonl(test_path)
69
+
70
+ logger.info(f"Train: {len(train_data)} samples, Test: {len(test_data)} samples")
71
+
72
+ train_dataset = Dataset.from_list(train_data)
73
+ test_dataset = Dataset.from_list(test_data)
74
+
75
+ # ---------------------------------------------------------------
76
+ # 2. Tokenize with label alignment
77
+ # ---------------------------------------------------------------
78
+ model_name = args.model_name
79
+ tokenizer = AutoTokenizer.from_pretrained(model_name)
80
+
81
+ def tokenize_and_align_labels(examples):
82
+ """
83
+ Tokenize text and realign NER labels with sub-word tokens.
84
+ Sub-word continuations receive label -100 (ignored by loss function).
85
+ """
86
+ tokenized_inputs = tokenizer(
87
+ examples["tokens"],
88
+ truncation=True,
89
+ padding="max_length",
90
+ max_length=128,
91
+ is_split_into_words=True
92
+ )
93
+
94
+ all_labels = []
95
+ for i, labels in enumerate(examples["ner_tags"]):
96
+ word_ids = tokenized_inputs.word_ids(batch_index=i)
97
+ label_ids = []
98
+ previous_word_idx = None
99
+ for word_idx in word_ids:
100
+ if word_idx is None:
101
+ label_ids.append(-100) # Special tokens
102
+ elif word_idx != previous_word_idx:
103
+ # Map WikiANN integer tags to our label space
104
+ tag = labels[word_idx] if word_idx < len(labels) else 0
105
+ label_ids.append(tag)
106
+ else:
107
+ # Sub-word continuation: use -100 or keep I- tag
108
+ tag = labels[word_idx] if word_idx < len(labels) else 0
109
+ # For sub-words of an entity, keep the I- variant
110
+ if tag in (1, 3, 5): # B-PER, B-ORG, B-LOC → convert to I-
111
+ tag = tag + 1
112
+ label_ids.append(tag)
113
+ previous_word_idx = word_idx
114
+ all_labels.append(label_ids)
115
+
116
+ tokenized_inputs["labels"] = all_labels
117
+ return tokenized_inputs
118
+
119
+ # Remove the "language" column if present
120
+ columns_to_remove = ["tokens", "ner_tags"]
121
+ if "language" in train_dataset.column_names:
122
+ columns_to_remove.append("language")
123
+
124
+ train_dataset = train_dataset.map(
125
+ tokenize_and_align_labels, batched=True, remove_columns=columns_to_remove
126
+ )
127
+ test_dataset = test_dataset.map(
128
+ tokenize_and_align_labels, batched=True, remove_columns=columns_to_remove
129
+ )
130
+
131
+ train_dataset.set_format("torch")
132
+ test_dataset.set_format("torch")
133
+
134
+ # ---------------------------------------------------------------
135
+ # 3. Load Model
136
+ # ---------------------------------------------------------------
137
+ model = AutoModelForTokenClassification.from_pretrained(
138
+ model_name,
139
+ num_labels=len(LABEL_LIST),
140
+ id2label=ID2LABEL,
141
+ label2id=LABEL2ID
142
+ )
143
+ logger.info(f"Loaded model: {model_name}")
144
+
145
+ # ---------------------------------------------------------------
146
+ # 4. Metrics (using seqeval for entity-level F1)
147
+ # ---------------------------------------------------------------
148
+ try:
149
+ from seqeval.metrics import f1_score as seqeval_f1, classification_report as seqeval_report
150
+ HAS_SEQEVAL = True
151
+ except ImportError:
152
+ logger.warning("seqeval not installed. Using token-level accuracy only.")
153
+ HAS_SEQEVAL = False
154
+
155
+ def compute_metrics(eval_pred):
156
+ logits, labels = eval_pred
157
+ predictions = np.argmax(logits, axis=-1)
158
+
159
+ if HAS_SEQEVAL:
160
+ # Convert to label strings, skipping -100
161
+ true_labels_seq = []
162
+ pred_labels_seq = []
163
+ for pred_seq, label_seq in zip(predictions, labels):
164
+ true_sent = []
165
+ pred_sent = []
166
+ for p, l in zip(pred_seq, label_seq):
167
+ if l != -100:
168
+ true_sent.append(ID2LABEL.get(int(l), "O"))
169
+ pred_sent.append(ID2LABEL.get(int(p), "O"))
170
+ true_labels_seq.append(true_sent)
171
+ pred_labels_seq.append(pred_sent)
172
+
173
+ f1 = seqeval_f1(true_labels_seq, pred_labels_seq, average="macro")
174
+ return {"f1_macro": f1}
175
+ else:
176
+ # Fallback: token-level accuracy
177
+ mask = labels != -100
178
+ correct = (predictions[mask] == labels[mask]).sum()
179
+ total = mask.sum()
180
+ return {"accuracy": float(correct / total)}
181
+
182
+ # ---------------------------------------------------------------
183
+ # 5. Training Configuration
184
+ # ---------------------------------------------------------------
185
+ output_dir = args.output_dir
186
+ data_collator = DataCollatorForTokenClassification(tokenizer)
187
+
188
+ training_args = TrainingArguments(
189
+ output_dir=output_dir,
190
+ num_train_epochs=args.epochs,
191
+ per_device_train_batch_size=args.batch_size,
192
+ per_device_eval_batch_size=args.batch_size * 2,
193
+ learning_rate=args.learning_rate,
194
+ weight_decay=0.01,
195
+ eval_strategy="epoch",
196
+ save_strategy="epoch",
197
+ load_best_model_at_end=True,
198
+ metric_for_best_model="f1_macro" if HAS_SEQEVAL else "accuracy",
199
+ greater_is_better=True,
200
+ save_total_limit=2,
201
+ logging_steps=100,
202
+ warmup_ratio=0.1,
203
+ fp16=args.fp16,
204
+ report_to="none",
205
+ )
206
+
207
+ trainer = Trainer(
208
+ model=model,
209
+ args=training_args,
210
+ train_dataset=train_dataset,
211
+ eval_dataset=test_dataset,
212
+ data_collator=data_collator,
213
+ compute_metrics=compute_metrics,
214
+ callbacks=[EarlyStoppingCallback(early_stopping_patience=2)]
215
+ )
216
+
217
+ # ---------------------------------------------------------------
218
+ # 6. Train
219
+ # ---------------------------------------------------------------
220
+ logger.info("Starting NER training...")
221
+ trainer.train()
222
+
223
+ # ---------------------------------------------------------------
224
+ # 7. Evaluate
225
+ # ---------------------------------------------------------------
226
+ results = trainer.evaluate()
227
+ logger.info(f"Evaluation results: {results}")
228
+
229
+ if HAS_SEQEVAL:
230
+ predictions = trainer.predict(test_dataset)
231
+ pred_labels = np.argmax(predictions.predictions, axis=-1)
232
+ true_labels = predictions.label_ids
233
+
234
+ true_labels_seq = []
235
+ pred_labels_seq = []
236
+ for pred_seq, label_seq in zip(pred_labels, true_labels):
237
+ true_sent = []
238
+ pred_sent = []
239
+ for p, l in zip(pred_seq, label_seq):
240
+ if l != -100:
241
+ true_sent.append(ID2LABEL.get(int(l), "O"))
242
+ pred_sent.append(ID2LABEL.get(int(p), "O"))
243
+ true_labels_seq.append(true_sent)
244
+ pred_labels_seq.append(pred_sent)
245
+
246
+ report = seqeval_report(true_labels_seq, pred_labels_seq)
247
+ logger.info(f"\nEntity-level Classification Report:\n{report}")
248
+
249
+ # ---------------------------------------------------------------
250
+ # 8. Save
251
+ # ---------------------------------------------------------------
252
+ trainer.save_model(output_dir)
253
+ tokenizer.save_pretrained(output_dir)
254
+ logger.info(f"Model saved to {output_dir}")
255
+
256
+ if args.push_to_hub:
257
+ trainer.push_to_hub(args.push_to_hub)
258
+
259
+
260
+ if __name__ == "__main__":
261
+ parser = argparse.ArgumentParser(description="Fine-tune a NER model")
262
+ parser.add_argument("--model-name", default="xlm-roberta-base", help="Base model name")
263
+ parser.add_argument("--data-dir", default="nlp_training/data", help="Directory with prepared datasets")
264
+ parser.add_argument("--output-dir", default="models/ner_model", help="Output directory")
265
+ parser.add_argument("--epochs", type=int, default=5, help="Number of training epochs")
266
+ parser.add_argument("--batch-size", type=int, default=16, help="Training batch size")
267
+ parser.add_argument("--learning-rate", type=float, default=3e-5, help="Learning rate")
268
+ parser.add_argument("--fp16", action="store_true", help="Use mixed precision training")
269
+ parser.add_argument("--push-to-hub", type=str, default=None, help="HuggingFace Hub repo")
270
+ args = parser.parse_args()
271
+ main(args)
nlp_training/train_sentiment_analyzer.py ADDED
@@ -0,0 +1,195 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Fine-tune a multilingual sentiment analyzer for WhatsNew News Intelligence Engine.
3
+
4
+ Model: XLM-RoBERTa-base (handles Arabic + English in a single model)
5
+ Dataset: ASAD (Arabic tweets) + SST-2 (English sentences) → 3 sentiment labels
6
+
7
+ Usage:
8
+ # 1. First prepare the data:
9
+ python nlp_training/prepare_datasets.py --task sentiment
10
+
11
+ # 2. Then fine-tune:
12
+ python nlp_training/train_sentiment_analyzer.py
13
+
14
+ # 3. The fine-tuned model will be saved to ./models/sentiment_analyzer/
15
+
16
+ Requirements:
17
+ pip install transformers datasets torch scikit-learn accelerate
18
+ """
19
+
20
+ import os
21
+ import json
22
+ import logging
23
+ import argparse
24
+ import numpy as np
25
+
26
+ logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
27
+ logger = logging.getLogger(__name__)
28
+
29
+ # Sentiment labels
30
+ LABEL_LIST = ["Positive", "Negative", "Neutral"]
31
+ LABEL2ID = {label: i for i, label in enumerate(LABEL_LIST)}
32
+ ID2LABEL = {i: label for i, label in enumerate(LABEL_LIST)}
33
+
34
+
35
+ def load_jsonl(filepath):
36
+ samples = []
37
+ with open(filepath, 'r', encoding='utf-8') as f:
38
+ for line in f:
39
+ samples.append(json.loads(line.strip()))
40
+ return samples
41
+
42
+
43
+ def main(args):
44
+ from transformers import (
45
+ AutoTokenizer,
46
+ AutoModelForSequenceClassification,
47
+ TrainingArguments,
48
+ Trainer,
49
+ EarlyStoppingCallback
50
+ )
51
+ from datasets import Dataset
52
+ from sklearn.metrics import accuracy_score, f1_score, classification_report
53
+
54
+ # ---------------------------------------------------------------
55
+ # 1. Load prepared datasets
56
+ # ---------------------------------------------------------------
57
+ data_dir = args.data_dir
58
+ train_path = os.path.join(data_dir, "sentiment_train.jsonl")
59
+ test_path = os.path.join(data_dir, "sentiment_test.jsonl")
60
+
61
+ if not os.path.exists(train_path):
62
+ logger.error(f"Training data not found at {train_path}. Run prepare_datasets.py first.")
63
+ return
64
+
65
+ train_data = load_jsonl(train_path)
66
+ test_data = load_jsonl(test_path)
67
+
68
+ # Filter to only include our known labels
69
+ train_data = [s for s in train_data if s["label"] in LABEL2ID]
70
+ test_data = [s for s in test_data if s["label"] in LABEL2ID]
71
+
72
+ logger.info(f"Train: {len(train_data)} samples, Test: {len(test_data)} samples")
73
+
74
+ # Label distribution
75
+ from collections import Counter
76
+ train_dist = Counter(s["label"] for s in train_data)
77
+ logger.info(f"Train label distribution: {dict(train_dist)}")
78
+
79
+ # Convert to HuggingFace Dataset
80
+ train_dataset = Dataset.from_list([
81
+ {"text": s["text"], "label": LABEL2ID[s["label"]]} for s in train_data
82
+ ])
83
+ test_dataset = Dataset.from_list([
84
+ {"text": s["text"], "label": LABEL2ID[s["label"]]} for s in test_data
85
+ ])
86
+
87
+ # ---------------------------------------------------------------
88
+ # 2. Tokenize
89
+ # ---------------------------------------------------------------
90
+ model_name = args.model_name
91
+ tokenizer = AutoTokenizer.from_pretrained(model_name)
92
+
93
+ def tokenize_fn(examples):
94
+ return tokenizer(examples["text"], truncation=True, padding="max_length", max_length=128)
95
+
96
+ train_dataset = train_dataset.map(tokenize_fn, batched=True, remove_columns=["text"])
97
+ test_dataset = test_dataset.map(tokenize_fn, batched=True, remove_columns=["text"])
98
+
99
+ train_dataset.set_format("torch")
100
+ test_dataset.set_format("torch")
101
+
102
+ # ---------------------------------------------------------------
103
+ # 3. Load Model
104
+ # ---------------------------------------------------------------
105
+ model = AutoModelForSequenceClassification.from_pretrained(
106
+ model_name,
107
+ num_labels=len(LABEL_LIST),
108
+ id2label=ID2LABEL,
109
+ label2id=LABEL2ID
110
+ )
111
+ logger.info(f"Loaded model: {model_name}")
112
+
113
+ # ---------------------------------------------------------------
114
+ # 4. Training Configuration
115
+ # ---------------------------------------------------------------
116
+ output_dir = args.output_dir
117
+
118
+ training_args = TrainingArguments(
119
+ output_dir=output_dir,
120
+ num_train_epochs=args.epochs,
121
+ per_device_train_batch_size=args.batch_size,
122
+ per_device_eval_batch_size=args.batch_size * 2,
123
+ learning_rate=args.learning_rate,
124
+ weight_decay=0.01,
125
+ eval_strategy="epoch",
126
+ save_strategy="epoch",
127
+ load_best_model_at_end=True,
128
+ metric_for_best_model="f1_macro",
129
+ greater_is_better=True,
130
+ save_total_limit=2,
131
+ logging_steps=100,
132
+ warmup_ratio=0.1,
133
+ fp16=args.fp16,
134
+ report_to="none",
135
+ )
136
+
137
+ def compute_metrics(eval_pred):
138
+ logits, labels = eval_pred
139
+ predictions = np.argmax(logits, axis=-1)
140
+ acc = accuracy_score(labels, predictions)
141
+ f1 = f1_score(labels, predictions, average="macro")
142
+ return {"accuracy": acc, "f1_macro": f1}
143
+
144
+ trainer = Trainer(
145
+ model=model,
146
+ args=training_args,
147
+ train_dataset=train_dataset,
148
+ eval_dataset=test_dataset,
149
+ compute_metrics=compute_metrics,
150
+ callbacks=[EarlyStoppingCallback(early_stopping_patience=2)]
151
+ )
152
+
153
+ # ---------------------------------------------------------------
154
+ # 5. Train
155
+ # ---------------------------------------------------------------
156
+ logger.info("Starting sentiment training...")
157
+ trainer.train()
158
+
159
+ # ---------------------------------------------------------------
160
+ # 6. Evaluate with detailed report
161
+ # ---------------------------------------------------------------
162
+ results = trainer.evaluate()
163
+ logger.info(f"Evaluation results: {results}")
164
+
165
+ # Detailed per-class report
166
+ predictions = trainer.predict(test_dataset)
167
+ pred_labels = np.argmax(predictions.predictions, axis=-1)
168
+ true_labels = predictions.label_ids
169
+ report = classification_report(true_labels, pred_labels, target_names=LABEL_LIST)
170
+ logger.info(f"\nClassification Report:\n{report}")
171
+
172
+ # ---------------------------------------------------------------
173
+ # 7. Save
174
+ # ---------------------------------------------------------------
175
+ trainer.save_model(output_dir)
176
+ tokenizer.save_pretrained(output_dir)
177
+ logger.info(f"Model saved to {output_dir}")
178
+
179
+ if args.push_to_hub:
180
+ trainer.push_to_hub(args.push_to_hub)
181
+ logger.info(f"Model pushed to HuggingFace Hub: {args.push_to_hub}")
182
+
183
+
184
+ if __name__ == "__main__":
185
+ parser = argparse.ArgumentParser(description="Fine-tune a sentiment analyzer")
186
+ parser.add_argument("--model-name", default="xlm-roberta-base", help="Base model name")
187
+ parser.add_argument("--data-dir", default="nlp_training/data", help="Directory with prepared datasets")
188
+ parser.add_argument("--output-dir", default="models/sentiment_analyzer", help="Output directory")
189
+ parser.add_argument("--epochs", type=int, default=4, help="Number of training epochs")
190
+ parser.add_argument("--batch-size", type=int, default=32, help="Training batch size")
191
+ parser.add_argument("--learning-rate", type=float, default=2e-5, help="Learning rate")
192
+ parser.add_argument("--fp16", action="store_true", help="Use mixed precision training")
193
+ parser.add_argument("--push-to-hub", type=str, default=None, help="HuggingFace Hub repo")
194
+ args = parser.parse_args()
195
+ main(args)
requirements.txt ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ annotated-types==0.8.0
2
+ anyio==4.14.2
3
+ beautifulsoup4==4.15.0
4
+ CacheControl==0.14.4
5
+ certifi==2026.7.22
6
+ cffi==2.1.1
7
+ charset-normalizer==3.4.9
8
+ cryptography==50.0.0
9
+ feedparser==6.0.14
10
+ feedparser-sgmllib==2.1.0
11
+ firebase_admin==7.5.0
12
+ google-api-core==2.33.0
13
+ google-auth==2.56.2
14
+ google-cloud-core==2.6.0
15
+ google-cloud-firestore==2.28.0
16
+ google-cloud-storage==3.13.0
17
+ google-crc32c==1.8.0
18
+ google-resumable-media==2.10.0
19
+ googleapis-common-protos==1.75.0
20
+ grpcio==1.83.0
21
+ grpcio-status==1.83.0
22
+ h11==0.16.0
23
+ h2==4.4.1
24
+ hpack==4.2.0
25
+ httpcore==1.0.9
26
+ httpx==0.28.1
27
+ hyperframe==6.1.0
28
+ idna==3.18
29
+ msgpack==1.2.1
30
+ proto-plus==1.28.2
31
+ protobuf==7.35.1
32
+ pyasn1==0.6.4
33
+ pyasn1_modules==0.4.2
34
+ pycparser==3.0
35
+ pydantic==2.13.4
36
+ pydantic_core==2.46.4
37
+ PyJWT==2.13.0
38
+ requests==2.34.2
39
+ schedule==1.2.2
40
+ soupsieve==2.9.1
41
+ typing-inspection==0.4.2
42
+ typing_extensions==4.16.0
43
+ urllib3==2.7.0
44
+
45
+ # NLP & Machine Learning Dependencies
46
+ transformers
47
+ datasets
48
+ scikit-learn
49
+ accelerate
50
+ seqeval
51
+ keybert
52
+ sentence-transformers
53
+ numpy
54
+ gradio
requirements_space.txt ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ gradio>=4.0.0
2
+ transformers
3
+ torch
4
+ datasets
5
+ keybert
6
+ sentence-transformers
7
+ beautifulsoup4
8
+ requests
9
+ pydantic
scrapers/__pycache__/rss_scraper.cpython-314.pyc ADDED
Binary file (7.82 kB). View file
 
scrapers/base_scraper.py ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import requests
2
+ from bs4 import BeautifulSoup
3
+ import logging
4
+ from typing import Optional
5
+
6
+ logger = logging.getLogger(__name__)
7
+
8
+ class BaseScraper:
9
+ headers = {
10
+ 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/114.0.0.0 Safari/537.36'
11
+ }
12
+
13
+ def fetch_html(self, url: str) -> Optional[BeautifulSoup]:
14
+ """Fetches the HTML content of the given URL and returns a BeautifulSoup object."""
15
+ try:
16
+ response = requests.get(url, headers=self.headers, timeout=10)
17
+ response.raise_for_status()
18
+ return BeautifulSoup(response.content, 'html.parser')
19
+ except Exception as e:
20
+ logger.error(f"Failed to fetch {url}: {e}")
21
+ return None
scrapers/rss_scraper.py ADDED
@@ -0,0 +1,128 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import datetime
2
+ import logging
3
+ import feedparser
4
+ from typing import List, Optional
5
+ from bs4 import BeautifulSoup
6
+ from domain.entities.article import Article
7
+ from domain.interfaces import ScraperInterface
8
+ import requests
9
+ import time
10
+
11
+ # Set a custom user agent for feedparser to bypass basic bot protections
12
+ feedparser.USER_AGENT = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/115.0.0.0 Safari/537.36"
13
+
14
+ logger = logging.getLogger(__name__)
15
+
16
+ class RssScraper(ScraperInterface):
17
+ def __init__(self, source_name: str, language: str, feed_url: str, category: str = "World"):
18
+ self.source_name = source_name
19
+ self.language = language
20
+ self.feed_url = feed_url
21
+ self.category = category
22
+
23
+ def scrape(self) -> List[Article]:
24
+ articles = []
25
+ try:
26
+ feed = feedparser.parse(self.feed_url)
27
+
28
+ # Check for network or parsing errors that feedparser hides
29
+ if getattr(feed, 'bozo', 0) == 1 and hasattr(feed, 'bozo_exception'):
30
+ logger.warning(f"Feed error for {self.feed_url}: {feed.bozo_exception}")
31
+ if not feed.entries:
32
+ return articles
33
+
34
+ for entry in feed.entries[:10]: # Limit to top 10 per feed
35
+ title = entry.get("title", "")
36
+ link = entry.get("link", "")
37
+
38
+ # Get summary, remove HTML tags
39
+ summary_raw = entry.get("description", "")
40
+ summary = BeautifulSoup(summary_raw, "html.parser").get_text()[:200]
41
+
42
+ if not title or not link:
43
+ continue
44
+
45
+ # Try to extract image (from RSS or OpenGraph meta tags)
46
+ image_url = self._extract_image(entry, link=link)
47
+
48
+ # Parse date or default to now
49
+ try:
50
+ # feedparser parses date into struct_time
51
+ published_at = datetime.datetime.fromtimestamp(time.mktime(entry.published_parsed)) if hasattr(entry, 'published_parsed') and entry.published_parsed else datetime.datetime.now()
52
+ except Exception:
53
+ published_at = datetime.datetime.now()
54
+
55
+ article = Article(
56
+ id=Article.generate_id_from_url(link),
57
+ title=title,
58
+ source=self.source_name,
59
+ url=link,
60
+ imageUrl=image_url,
61
+ hasImage=True if image_url else False,
62
+ summary=summary + "..." if len(summary) >= 200 else summary,
63
+ publishedAt=published_at,
64
+ language=self.language,
65
+ category=self.category
66
+ )
67
+ articles.append(article)
68
+ except Exception as e:
69
+ logger.error(f"Failed to scrape RSS {self.feed_url}: {e}")
70
+
71
+ return articles
72
+
73
+ def _extract_image(self, entry, link: str = None) -> Optional[str]:
74
+ # 1. Check media_content
75
+ if hasattr(entry, 'media_content') and entry.media_content:
76
+ for media in entry.media_content:
77
+ if media.get('url'):
78
+ return media.get('url')
79
+
80
+ # 2. Check media_thumbnail
81
+ if hasattr(entry, 'media_thumbnail') and entry.media_thumbnail:
82
+ return entry.media_thumbnail[0].get('url')
83
+
84
+ # 3. Check enclosures
85
+ if hasattr(entry, 'enclosures') and entry.enclosures:
86
+ for enc in entry.enclosures:
87
+ if 'image' in enc.get('type', '') and enc.get('href'):
88
+ return enc.get('href')
89
+
90
+ # 4. Check links for image type
91
+ if hasattr(entry, 'links'):
92
+ for l in entry.links:
93
+ if 'image' in l.get('type', ''):
94
+ return l.get('href')
95
+
96
+ # 5. Check description/content HTML for <img> tags
97
+ for field in ['description', 'content']:
98
+ if hasattr(entry, field):
99
+ content_list = getattr(entry, field)
100
+ raw_html = content_list[0].get('value', '') if isinstance(content_list, list) else content_list
101
+ soup = BeautifulSoup(str(raw_html), "html.parser")
102
+ img = soup.find('img')
103
+ if img and img.get('src'):
104
+ return img.get('src')
105
+
106
+ # 6. Layer 2: OpenGraph Web Scraping Fallback (Fast 2s timeout)
107
+ if link and link.startswith('http'):
108
+ try:
109
+ headers = {"User-Agent": feedparser.USER_AGENT}
110
+ resp = requests.get(link, headers=headers, timeout=2.5)
111
+ if resp.status_code == 200:
112
+ soup = BeautifulSoup(resp.text, "html.parser")
113
+ # Check og:image, twitter:image, image_src
114
+ og_img = soup.find('meta', property='og:image') or soup.find('meta', attrs={'name': 'og:image'})
115
+ if og_img and og_img.get('content'):
116
+ return og_img.get('content')
117
+
118
+ tw_img = soup.find('meta', attrs={'name': 'twitter:image'}) or soup.find('meta', property='twitter:image')
119
+ if tw_img and tw_img.get('content'):
120
+ return tw_img.get('content')
121
+
122
+ rel_img = soup.find('link', rel='image_src')
123
+ if rel_img and rel_img.get('href'):
124
+ return rel_img.get('href')
125
+ except Exception:
126
+ pass
127
+
128
+ return None
services/__pycache__/ai_classifier.cpython-314.pyc ADDED
Binary file (11.7 kB). View file
 
services/__pycache__/duplicate_detector.cpython-314.pyc ADDED
Binary file (2.82 kB). View file
 
services/ai_classifier.py ADDED
@@ -0,0 +1,185 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import requests
2
+ import json
3
+ import os
4
+ import re
5
+ import logging
6
+ from domain.entities.article import Article
7
+
8
+ from services.lexicons import (
9
+ POSITIVE_WORDS,
10
+ NEGATIVE_WORDS,
11
+ CATEGORIES_MAP,
12
+ LOCATIONS_MAP,
13
+ MIDDLE_EAST_COUNTRIES,
14
+ STOP_WORDS
15
+ )
16
+ from services.local_nlp_classifier import LocalNLPClassifier
17
+
18
+ logger = logging.getLogger(__name__)
19
+
20
+ class HuggingFaceClassifier:
21
+ def __init__(self, api_key: str = None):
22
+ self.api_key = api_key or os.environ.get("HF_API_KEY")
23
+ # List of models to try in sequence if one hits rate limits or credit depletion
24
+ self.fallback_models = [
25
+ "Qwen/Qwen2.5-7B-Instruct",
26
+ "Qwen/Qwen2.5-Coder-7B-Instruct",
27
+ "deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B",
28
+ "meta-llama/Llama-3.1-8B-Instruct"
29
+ ]
30
+ self.api_url = "https://router.huggingface.co/v1/chat/completions"
31
+ # Tier 2: Local NLP transformer models (loaded lazily on first use)
32
+ self._local_nlp = LocalNLPClassifier()
33
+
34
+ def enrich_article(self, article: Article) -> Article:
35
+ if not self.api_key:
36
+ logger.warning("No HF_API_KEY provided. Using local heuristic enrichment.")
37
+ return self._local_heuristic_fallback(article)
38
+
39
+ headers = {
40
+ "Authorization": f"Bearer {self.api_key}",
41
+ "Content-Type": "application/json"
42
+ }
43
+
44
+ prompt = """You are an article metadata enrichment service. Analyze the provided article title and summary and return ONLY valid JSON. Do not include any explanations or conversational text. Infer a single high-level category (Politics, Business, Technology, AI, Science, Health, Sports, Entertainment, Gaming, Culture, Education, Economy, Crime, Military, Weather, Environment, Travel, Lifestyle, Religion, Opinion, Local News, World), an optional subcategory, the primary region, country, and city if identifiable, 5–15 normalized tags, important keywords, named entities grouped into people, organizations, and locations, a sentiment (Positive, Negative, Neutral, Mixed), an importance score from 1 to 10 based on global significance, and an estimated readingTime in minutes. Do not invent facts that are not reasonably inferable from the title and summary. If information cannot be determined confidently, return null or an empty array. Return ONLY JSON matching this exact schema:
45
+ {
46
+ "category": "String",
47
+ "subcategory": "String",
48
+ "region": "String",
49
+ "country": "String",
50
+ "city": "String",
51
+ "tags": [],
52
+ "keywords": [],
53
+ "people": [],
54
+ "organizations": [],
55
+ "locations": [],
56
+ "sentiment": "String",
57
+ "importance": 0,
58
+ "readingTime": 0
59
+ }"""
60
+
61
+ # Try models in order until one succeeds
62
+ for model in self.fallback_models:
63
+ payload = {
64
+ "model": model,
65
+ "messages": [
66
+ {"role": "system", "content": prompt},
67
+ {"role": "user", "content": f"Title: {article.title}\nSummary: {article.summary}"}
68
+ ],
69
+ "max_tokens": 1000,
70
+ "temperature": 0.1
71
+ }
72
+
73
+ try:
74
+ response = requests.post(self.api_url, headers=headers, json=payload, timeout=20)
75
+ if response.status_code == 200:
76
+ result = response.json()
77
+ content = result['choices'][0]['message']['content'].strip()
78
+
79
+ if content.startswith("```json"):
80
+ content = content[7:-3].strip()
81
+ elif content.startswith("```"):
82
+ content = content[3:-3].strip()
83
+
84
+ data = json.loads(content)
85
+
86
+ article.category = data.get('category') or article.category or "General"
87
+ article.subcategory = data.get('subcategory')
88
+ article.region = data.get('region')
89
+ article.country = data.get('country')
90
+ article.city = data.get('city')
91
+ article.tags = data.get('tags') or []
92
+ article.keywords = data.get('keywords') or []
93
+ article.people = data.get('people') or []
94
+ article.organizations = data.get('organizations') or []
95
+ article.locations = data.get('locations') or []
96
+ article.sentiment = data.get('sentiment') or "Neutral"
97
+ article.importance = data.get('importance') or 5
98
+ article.readingTime = data.get('readingTime') or self._estimate_reading_time(article)
99
+
100
+ return article
101
+ else:
102
+ logger.warning(f"HF Model {model} failed with status {response.status_code}. Trying next fallback...")
103
+ except Exception as e:
104
+ logger.warning(f"Exception trying model {model}: {e}. Trying next fallback...")
105
+
106
+ # If all AI models failed, try Tier 2: local NLP transformer models
107
+ logger.warning(f"All HF API models failed for article '{article.title[:50]}'. Trying Tier 2 local NLP models...")
108
+ if self._local_nlp.is_available():
109
+ try:
110
+ enriched = self._local_nlp.enrich_article(article)
111
+ logger.info("Tier 2 local NLP enrichment succeeded.")
112
+ return enriched
113
+ except Exception as e:
114
+ logger.warning(f"Tier 2 local NLP failed: {e}")
115
+
116
+ # Tier 3: If all else failed, use lexicon-based heuristic fallback so fields are never null
117
+ logger.error(f"All enrichment tiers failed for article '{article.title[:50]}'. Falling back to Tier 3 heuristics.")
118
+ return self._local_heuristic_fallback(article)
119
+
120
+ def _estimate_reading_time(self, article: Article) -> int:
121
+ text = f"{article.title} {article.summary or ''}"
122
+ # Strip simple HTML if present
123
+ text = re.sub(r'<[^>]+>', '', text)
124
+ words = len(text.split())
125
+ return max(1, round(words / 180))
126
+
127
+ def _local_heuristic_fallback(self, article: Article) -> Article:
128
+ """High-precision rule-based NLP fallback using modular lexicons."""
129
+ full_text = f"{article.title} {article.summary or ''}".lower()
130
+
131
+ # 1. Category & Subcategory Detection
132
+ if not article.category or article.category in ["General", "Middle East", "Egypt", "World", "North Africa"]:
133
+ for cat_name, (keywords, subcat) in CATEGORIES_MAP.items():
134
+ if any(kw in full_text for kw in keywords):
135
+ article.category = cat_name
136
+ article.subcategory = subcat
137
+ break
138
+ if not article.category:
139
+ article.category = "General"
140
+
141
+ # 2. Entity & Location Extraction
142
+ detected_locations = []
143
+ for loc_kw, loc_name in LOCATIONS_MAP.items():
144
+ if loc_kw in full_text:
145
+ if loc_name not in detected_locations:
146
+ detected_locations.append(loc_name)
147
+
148
+ article.locations = detected_locations if detected_locations else ["Global"]
149
+ if detected_locations:
150
+ article.country = detected_locations[0]
151
+ article.region = "Middle East" if any(l in MIDDLE_EAST_COUNTRIES for l in detected_locations) else "Global"
152
+
153
+ # 3. Clean Keyword & Tag Extraction
154
+ clean_words = [
155
+ re.sub(r'[^\w\s]', '', w) for w in article.title.split()
156
+ if len(w) > 3 and w.lower() not in STOP_WORDS
157
+ ]
158
+
159
+ article.keywords = list(dict.fromkeys(clean_words[:6]))
160
+ article.tags = list(dict.fromkeys(clean_words[:10]))
161
+
162
+ # 4. Lexicon Sentiment Analysis
163
+ pos_count = sum(1 for w in POSITIVE_WORDS if w in full_text)
164
+ neg_count = sum(1 for w in NEGATIVE_WORDS if w in full_text)
165
+
166
+ if pos_count > neg_count:
167
+ article.sentiment = "Positive"
168
+ elif neg_count > pos_count:
169
+ article.sentiment = "Negative"
170
+ else:
171
+ article.sentiment = "Neutral"
172
+
173
+ # 5. Dynamic Importance Scoring (1 - 10)
174
+ high_impact_terms = ["عاجل", "قمة", "حرب", "وفاة", "رئيس", "كوارث", "breaking", "war", "president", "disaster", "crisis", "urgent"]
175
+ importance_score = 5
176
+ if any(term in full_text for term in high_impact_terms):
177
+ importance_score += 3
178
+ if len(detected_locations) > 1:
179
+ importance_score += 1
180
+ article.importance = min(10, importance_score)
181
+
182
+ # 6. Reading Time Estimation
183
+ article.readingTime = self._estimate_reading_time(article)
184
+
185
+ return article
services/duplicate_detector.py ADDED
@@ -0,0 +1,45 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import logging
2
+ from typing import List
3
+ from domain.entities.article import Article
4
+ from domain.repositories.article_repository import ArticleRepository
5
+
6
+ logger = logging.getLogger(__name__)
7
+
8
+ class DuplicateDetector:
9
+ def __init__(self, repository: ArticleRepository):
10
+ self.repository = repository
11
+
12
+ def filter_new_articles(self, articles: List[Article]) -> List[Article]:
13
+ """Robust multi-level duplicate detection (URL Hash, In-Batch Seen, Title Hash)."""
14
+ new_articles = []
15
+ seen_ids = set()
16
+ seen_title_hashes = set()
17
+ skipped_count = 0
18
+
19
+ for article in articles:
20
+ # 1. Normalize ID and Title Hash
21
+ article_id = article.id
22
+ title_hash = Article.generate_title_hash(article.title)
23
+
24
+ # 2. In-batch Duplicate Check (prevents duplicates inside the same scrape run)
25
+ if article_id in seen_ids or title_hash in seen_title_hashes:
26
+ logger.debug(f"In-batch duplicate detected: {article.title[:40]}")
27
+ skipped_count += 1
28
+ continue
29
+
30
+ # 3. Database Check (checks if article ID already exists in Firestore)
31
+ if self.repository.exists(article_id):
32
+ logger.debug(f"Article exists in Firestore: {article.title[:40]}")
33
+ skipped_count += 1
34
+ # Mark as seen so we don't check Firestore again if it reappears in feed
35
+ seen_ids.add(article_id)
36
+ seen_title_hashes.add(title_hash)
37
+ continue
38
+
39
+ # 4. Mark as unique and new
40
+ seen_ids.add(article_id)
41
+ seen_title_hashes.add(title_hash)
42
+ new_articles.append(article)
43
+
44
+ logger.info(f"Duplicate Detection complete: {len(new_articles)} new, {skipped_count} duplicates skipped.")
45
+ return new_articles
services/lexicons/__init__.py ADDED
@@ -0,0 +1,15 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from .sentiment_lexicon import POSITIVE_WORDS, NEGATIVE_WORDS
2
+ from .categories_lexicon import CATEGORIES_MAP
3
+ from .locations_lexicon import LOCATIONS_MAP, MIDDLE_EAST_COUNTRIES
4
+ from .stopwords_lexicon import STOP_WORDS
5
+ from .images_lexicon import CATEGORY_IMAGES_MAP
6
+
7
+ __all__ = [
8
+ "POSITIVE_WORDS",
9
+ "NEGATIVE_WORDS",
10
+ "CATEGORIES_MAP",
11
+ "LOCATIONS_MAP",
12
+ "MIDDLE_EAST_COUNTRIES",
13
+ "STOP_WORDS",
14
+ "CATEGORY_IMAGES_MAP"
15
+ ]
services/lexicons/__pycache__/__init__.cpython-314.pyc ADDED
Binary file (584 Bytes). View file
 
services/lexicons/__pycache__/categories_lexicon.cpython-314.pyc ADDED
Binary file (13.9 kB). View file
 
services/lexicons/__pycache__/images_lexicon.cpython-314.pyc ADDED
Binary file (2.25 kB). View file
 
services/lexicons/__pycache__/locations_lexicon.cpython-314.pyc ADDED
Binary file (7.37 kB). View file
 
services/lexicons/__pycache__/sentiment_lexicon.cpython-314.pyc ADDED
Binary file (8.25 kB). View file
 
services/lexicons/__pycache__/stopwords_lexicon.cpython-314.pyc ADDED
Binary file (3.32 kB). View file
 
services/lexicons/categories_lexicon.py ADDED
@@ -0,0 +1,135 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Extensive Category and Subcategory Keyword Mappings for Arabic & English News NLP
2
+
3
+ CATEGORIES_MAP = {
4
+ "Sports": (
5
+ [
6
+ "كرة", "مباراة", "دوري", "بطولة", "الأهلي", "الزمالك", "صلاح", "رياضة", "منتخب", "هداف",
7
+ "ملعب", "مدرب", "صفقة", "انتقالات", "كأس", "كرة قدم", "سلة", "كرة يد", "تنس", "أولمبياد",
8
+ "ريال مدريد", "برشلونة", "مانشستر", "ليفربول", "دوري أبطال", "كأس العالم", "رونالدو", "ميسي",
9
+ "football", "soccer", "match", "league", "cup", "nba", "tennis", "stadium", "coach",
10
+ "tournament", "player", "fifa", "uefa", "champion", "olympics", "nfl", "formula1", "f1",
11
+ "real madrid", "barcelona", "liverpool", "champions league", "world cup", "messi", "ronaldo"
12
+ ],
13
+ "Football"
14
+ ),
15
+ "Technology": (
16
+ [
17
+ "ذكاء اصطناعي", "تقنية", "هاتف", "تطبيق", "أبل", "سامسونج", "جوجل", "برمجيات", "روبوت",
18
+ "سحابة", "أمن سيبراني", "معالج", "آيفون", "أندرويد", "تحديث", "شاشة", "حاسوب", "ابتكار",
19
+ "شبكات", "5g", "رقائق", "سيليكون", "أنظمة", "خوارزمية", "شات جي بي تي", "إنترنت",
20
+ "ai", "artificial intelligence", "tech", "technology", "apple", "google", "software",
21
+ "cyber", "cybersecurity", "iphone", "android", "cloud", "robotics", "processor", "gadget",
22
+ "app", "application", "startup", "microsoft", "semiconductor", "chip", "chatgpt", "openai"
23
+ ],
24
+ "Tech News"
25
+ ),
26
+ "Gaming": (
27
+ [
28
+ "لعبة", "العاب", "ألعاب", "بلايستيشن", "إكس بوكس", "قيمينق", "جيمرز", "نينتندو", "ستيم",
29
+ "إي سبورتس", "مطور ألعاب", "تسريبات لعبة", "تحديث لعبة", "سوني", "مهمة", "بطولة العاب",
30
+ "ببجي", "فورتنايت", "كول اوف ديوتي", "فيفا 24", "العاب فيديو",
31
+ "game", "gaming", "playstation", "xbox", "nintendo", "esports", "steam", "gamer",
32
+ "ps5", "unreal engine", "graphics", "gameplay", "gta", "fortnite", "call of duty", "rpg", "roblox"
33
+ ],
34
+ "Gaming"
35
+ ),
36
+ "Business": (
37
+ [
38
+ "اقتصاد", "أسهم", "بورصة", "دولار", "تضخم", "شركات", "أرباح", "استثمار", "بنك", "مركزي",
39
+ "تجارة", "أسواق", "نفط", "غاز", "عملات", "عقارات", "استثمارات", "ميزانية", "صندوق النقد",
40
+ "فائدة", "بنك دولي", "استحواذ", "اندماج", "تداول", "ذهب", "سندات", "مشاريع",
41
+ "economy", "market", "markets", "stock", "stocks", "dollar", "inflation", "bank", "banking",
42
+ "finance", "financial", "investment", "investor", "revenue", "trade", "crypto", "bitcoin",
43
+ "real estate", "fed", "central bank", "gdp", "gold", "acquisition", "bonds"
44
+ ],
45
+ "Finance"
46
+ ),
47
+ "Health": (
48
+ [
49
+ "صحة", "مرض", "علاج", "طبيب", "أطباء", "مستشفى", "مستشفيات", "فيروس", "لقاح", "دواء",
50
+ "أدوية", "سرطان", "مناعة", "قلب", "حمية", "تغذية", "وباء", "منظمة الصحة", "عيادة",
51
+ "جراحة", "صحة نفسية", "أعراض", "تشخيص", "وقاية", "سمنة",
52
+ "health", "virus", "medical", "doctor", "hospital", "medicine", "vaccine", "cancer",
53
+ "disease", "epidemic", "pandemic", "wellness", "diet", "nutrition", "pharma", "who",
54
+ "surgery", "mental health", "symptoms", "diagnosis", "prevention"
55
+ ],
56
+ "Medical"
57
+ ),
58
+ "Science": (
59
+ [
60
+ "فضاء", "كوكب", "علماء", "مريخ", "ناسا", "بيئة", "مناخ", "أبحاث", "مجرة", "تلسكوب",
61
+ "احتباس حراري", "فيزياء", "كيمياء", "طبيعة", "احافير", "ذرة", "طاقة تجددة", "اكتشاف",
62
+ "علمي", "تجربة", "مختبر", "أحياء", "جينات", "وراثة",
63
+ "science", "space", "nasa", "planet", "climate", "research", "galaxy", "telescope",
64
+ "astronomy", "physics", "biology", "fossil", "atom", "renewable energy", "nature",
65
+ "discovery", "scientific", "laboratory", "genetics", "dna"
66
+ ],
67
+ "Space & Research"
68
+ ),
69
+ "Food": (
70
+ [
71
+ "وجبة", "طعام", "وصفة", "وصفات", "مطعم", "مطاعم", "طهي", "أكل", "مطبخ", "شيف", "عشاء",
72
+ "غداء", "إفطار", "مكونات", "طبق", "حلويات", "مخبوزات", "طبخ", "حلويات شرقية", "مشروبات",
73
+ "food", "recipe", "recipes", "cooking", "kitchen", "dish", "dishes", "restaurant",
74
+ "chef", "meal", "cuisine", "flavor", "bakery", "dessert", "dining", "flavor", "baking"
75
+ ],
76
+ "Culinary"
77
+ ),
78
+ "Entertainment": (
79
+ [
80
+ "فيلم", "أفلام", "مسلسل", "مسلسلات", "سينما", "موسيقى", "فنان", "فنانة", "مهرجان", "أغنية",
81
+ "هوليوود", "دراما", "كوميديا", "شباك التذاكر", "أوسكار", "عرض أول", "حفلة", "أغاني",
82
+ "تريند", "مشاهير", "نجوم", "بوستر", "تريلر", "برومو",
83
+ "movie", "movies", "film", "cinema", "music", "actor", "actress", "hollywood", "series",
84
+ "tv show", "box office", "oscars", "grammy", "concert", "celebrity", "drama", "trailer", "trend"
85
+ ],
86
+ "Movies & TV"
87
+ ),
88
+ "Politics": (
89
+ [
90
+ "رئيس", "وزير", "وزراء", "حكومة", "برلمان", "انتخابات", "سفارة", "سفير", "دبلوماسية",
91
+ "مفاوضات", "معاهدة", "مجلس الأمن", "الأمم المتحدة", "حزب", "سياسة", "خارجية", "دفاع",
92
+ "عسكري", "جيش", "قوات", "سياسي", "قمة", "مؤتمر صحفي", "سياسيون",
93
+ "president", "minister", "government", "parliament", "election", "elections", "diplomacy",
94
+ "diplomatic", "ambassador", "treaty", "un security council", "senate", "congress", "policy",
95
+ "military", "army", "forces", "summit", "geopolitics"
96
+ ],
97
+ "Government"
98
+ ),
99
+ "Education": (
100
+ [
101
+ "تعليم", "مدرسة", "مدارس", "جامعة", "جامعات", "طلاب", "طالب", "امتحانات", "امتحان",
102
+ "نتيجة", "تنسيق", "وزارة التربية والتعليم", "بكالوريا", "ثانوية عامة", "دراسة", "منحة",
103
+ "education", "school", "schools", "university", "student", "students", "exam", "exams",
104
+ "degree", "scholarship", "academy", "curriculum", "campus"
105
+ ],
106
+ "Education"
107
+ ),
108
+ "Crime": (
109
+ [
110
+ "جريمة", "جرائم", "شرطة", "قبض", "اعتقال", "مخدرات", "تحقيق", "نيابة", "محكمة", "قضاء",
111
+ "احتيال", "سرقة", "سطو", "قتل", "عصابة", "تهريب", "قاضي",
112
+ "crime", "police", "arrest", "arrested", "drugs", "investigation", "court", "judge",
113
+ "fraud", "theft", "robbery", "murder", "gang", "smuggling", "trial"
114
+ ],
115
+ "Crime & Justice"
116
+ ),
117
+ "Weather": (
118
+ [
119
+ "طقس", "مناخ", "حرارة", "أمطار", "مطر", "عاصفة", "عواصف", "ثلوج", "رياح", "أرصاد",
120
+ "موجة حارة", "إعصار", "سيول", "درجات الحرارة",
121
+ "weather", "forecast", "temperature", "rain", "storm", "snow", "wind", "hurricane",
122
+ "heatwave", "blizzard", "meteorology"
123
+ ],
124
+ "Weather & Climate"
125
+ ),
126
+ "Travel": (
127
+ [
128
+ "سفر", "سياحة", "فندق", "فنادق", "طيران", "رحلة", "رحلات", "وجهة", "تأشيرة", "فيزا",
129
+ "مطار", "معالم", "شاطئ", "منتجع",
130
+ "travel", "tourism", "hotel", "hotels", "flight", "flights", "destination", "visa",
131
+ "airport", "beach", "resort", "vacation"
132
+ ],
133
+ "Travel & Leisure"
134
+ )
135
+ }
services/lexicons/images_lexicon.py ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # High-Resolution Category Fallback Cover Images (Royalty-Free Unsplash Assets)
2
+
3
+ CATEGORY_IMAGES_MAP = {
4
+ "Technology": "https://images.unsplash.com/photo-1518770660439-4636190af475?auto=format&fit=crop&w=1000&q=80",
5
+ "Gaming": "https://images.unsplash.com/photo-1538481199705-c710c4e965fc?auto=format&fit=crop&w=1000&q=80",
6
+ "Sports": "https://images.unsplash.com/photo-1461896836934-ffe607ba8211?auto=format&fit=crop&w=1000&q=80",
7
+ "Business": "https://images.unsplash.com/photo-1611974789855-9c2a0a7236a3?auto=format&fit=crop&w=1000&q=80",
8
+ "Health": "https://images.unsplash.com/photo-1505751172876-fa1923c5c528?auto=format&fit=crop&w=1000&q=80",
9
+ "Science": "https://images.unsplash.com/photo-1451187580459-43490279c0fa?auto=format&fit=crop&w=1000&q=80",
10
+ "Food": "https://images.unsplash.com/photo-1504674900247-0877df9cc836?auto=format&fit=crop&w=1000&q=80",
11
+ "Entertainment": "https://images.unsplash.com/photo-1489599849927-2ee91cede3ba?auto=format&fit=crop&w=1000&q=80",
12
+ "Politics": "https://images.unsplash.com/photo-1541872703-74c5e44368f9?auto=format&fit=crop&w=1000&q=80",
13
+ "Education": "https://images.unsplash.com/photo-1523240795612-9a054b0db644?auto=format&fit=crop&w=1000&q=80",
14
+ "Crime": "https://images.unsplash.com/photo-1589829545856-d10d557cf95f?auto=format&fit=crop&w=1000&q=80",
15
+ "Weather": "https://images.unsplash.com/photo-1516912481808-3406841bd33c?auto=format&fit=crop&w=1000&q=80",
16
+ "Travel": "https://images.unsplash.com/photo-1488646953014-85cb44e25828?auto=format&fit=crop&w=1000&q=80",
17
+ "Middle East": "https://images.unsplash.com/photo-1512453979798-5ea266f8880c?auto=format&fit=crop&w=1000&q=80",
18
+ "Egypt": "https://images.unsplash.com/photo-1572252821128-d53b16d39d5e?auto=format&fit=crop&w=1000&q=80",
19
+ "World": "https://images.unsplash.com/photo-1526778548025-fa2f459cd5c1?auto=format&fit=crop&w=1000&q=80",
20
+ "General": "https://images.unsplash.com/photo-1504711434969-e33886168f5c?auto=format&fit=crop&w=1000&q=80"
21
+ }
services/lexicons/locations_lexicon.py ADDED
@@ -0,0 +1,59 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Comprehensive Global and Middle East Geographic Gazetteer Mapping
2
+
3
+ LOCATIONS_MAP = {
4
+ # Egypt (Governorates & Cities)
5
+ "مصر": "Egypt", "القاهرة": "Cairo", "الإسكندرية": "Alexandria", "الجيزة": "Giza",
6
+ "سيناء": "Sinai", "شرم الشيخ": "Sharm El Sheikh", "العلمين": "El Alamein", "أسوان": "Aswan",
7
+ "الأقصر": "Luxor", "مطروح": "Matrouh", "بورسعيد": "Port Said", "السويس": "Suez",
8
+ "طنطا": "Tanta", "المنصورة": "Mansoura", "الزقازيق": "Zagazig", "الإسماعيلية": "Ismailia",
9
+ "الفيوم": "Faiyum", "بني سويف": "Beni Suef", "المنيا": "Minya", "أسيوط": "Asyut",
10
+ "سوهاج": "Sohag", "قنا": "Qena", "الغردقة": "Hurghada", "دمياط": "Damietta",
11
+ "egypt": "Egypt", "cairo": "Cairo", "alexandria": "Alexandria", "giza": "Giza",
12
+
13
+ # Palestine & Levant
14
+ "غزة": "Gaza", "فلسطين": "Palestine", "القدس": "Jerusalem", "الضفة": "West Bank",
15
+ "رام الله": "Ramallah", "رفح": "Rafah", "خان يونس": "Khan Younis", "جنين": "Jenin",
16
+ "نابلس": "Nablus", "بيروت": "Beirut", "لبنان": "Lebanon", "دمشق": "Damascus", "حلب": "Aleppo",
17
+ "سوريا": "Syria", "عمان": "Amman", "الأردن": "Jordan", "إربد": "Irbid", "الزرعاء": "Zarqaa",
18
+ "gaza": "Gaza", "palestine": "Palestine", "jerusalem": "Jerusalem", "lebanon": "Lebanon",
19
+ "syria": "Syria", "jordan": "Jordan", "beirut": "Beirut", "damascus": "Damascus",
20
+
21
+ # Gulf & Arabian Peninsula
22
+ "السعودية": "Saudi Arabia", "الرياض": "Riyadh", "جدة": "Jeddah", "مكة": "Mekkah", "المدينة": "Madinah",
23
+ "الدمام": "Dammam", "الخبر": "Khobar", "الإمارات": "UAE", "دبي": "Dubai", "أبوظبي": "Abu Dhabi",
24
+ "الشارقة": "Sharjah", "قطر": "Qatar", "الدوحة": "Doha", "الكويت": "Kuwait", "البحرين": "Bahrain",
25
+ "المنامة": "Manama", "عمان": "Oman", "مسقط": "Muscat", "اليمن": "Yemen", "صنعاء": "Sanaa", "عدن": "Aden",
26
+ "saudi arabia": "Saudi Arabia", "riyadh": "Riyadh", "dubai": "Dubai", "uae": "UAE",
27
+ "qatar": "Qatar", "doha": "Doha", "kuwait": "Kuwait", "abu dhabi": "Abu Dhabi",
28
+
29
+ # North Africa
30
+ "المغرب": "Morocco", "الرباط": "Rabat", "الدار البيضاء": "Casablanca", "مراكش": "Marrakech",
31
+ "فاس": "Fes", "طنجة": "Tangier", "الجزائر": "Algeria", "وهران": "Oran", "قسنطينة": "Constantine",
32
+ "تونس": "Tunisia", "صفاقس": "Sfax", "ليبيا": "Libya", "طرابلس": "Tripoli", "بنغازي": "Benghazi",
33
+ "السودان": "Sudan", "الخرطوم": "Khartoum", "أم درمان": "Omdurman",
34
+ "morocco": "Morocco", "algeria": "Algeria", "tunisia": "Tunisia", "libya": "Libya", "sudan": "Sudan",
35
+
36
+ # Iraq
37
+ "العراق": "Iraq", "بغداد": "Baghdad", "أربيل": "Erbil", "البصرة": "Basra", "الموصل": "Mosul",
38
+ "iraq": "Iraq", "baghdad": "Baghdad",
39
+
40
+ # Global Powers & Major Capitals
41
+ "واشنطن": "Washington", "أمريكا": "USA", "الولايات المتحدة": "USA", "نيويورك": "New York",
42
+ "شيكاغو": "Chicago", "كاليفورنيا": "California", "تكساس": "Texas", "فلوريدا": "Florida",
43
+ "باريس": "Paris", "فرنسا": "France", "مارسيليا": "Marseille", "لندن": "London", "بريطانيا": "UK",
44
+ "إنجلترا": "UK", "مانشستر": "Manchester", "برلين": "Berlin", "ألمانيا": "Germany", "ميونيخ": "Munich",
45
+ "بكين": "Beijing", "الصين": "China", "شنغهاي": "Shanghai", "موسكو": "Moscow", "روسيا": "Russia",
46
+ "طوكيو": "Tokyo", "اليابان": "Japan", "روما": "Rome", "إيطاليا": "Italy", "ميلانو": "Milan",
47
+ "مدريد": "Madrid", "إسبانيا": "Spain", "برشلونة": "Barcelona", "أنقرة": "Ankara", "تركيا": "Turkey",
48
+ "إسطنبول": "Istanbul", "كييف": "Kyiv", "أوكرانيا": "Ukraine", "سول": "Seoul", "كوريا": "Korea",
49
+ "نيودلهي": "New Delhi", "الهند": "India", "برازيليا": "Brasilia", "البرازيل": "Brazil",
50
+ "سدني": "Sydney", "أستراليا": "Australia", "تورونتو": "Toronto", "كندا": "Canada",
51
+ "usa": "USA", "us": "USA", "uk": "UK", "china": "China", "russia": "Russia", "france": "France",
52
+ "germany": "Germany", "japan": "Japan", "turkey": "Turkey", "ukraine": "Ukraine", "canada": "Canada"
53
+ }
54
+
55
+ MIDDLE_EAST_COUNTRIES = {
56
+ "Egypt", "Gaza", "Palestine", "Saudi Arabia", "UAE", "Qatar", "Kuwait", "Lebanon",
57
+ "Syria", "Jordan", "Iraq", "Yemen", "Oman", "Bahrain", "Sudan", "Libya", "Algeria",
58
+ "Morocco", "Tunisia"
59
+ }
services/lexicons/sentiment_lexicon.py ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Extensive sentiment lexicons for Arabic and English NLP classification
2
+
3
+ POSITIVE_WORDS = {
4
+ # Arabic Positive Cues
5
+ "فوز", "نجاح", "انتصار", "تألق", "تفوق", "نمو", "ازدهار", "أرباح", "ربح", "استقرار",
6
+ "اتفاق", "إنجاز", "افتتاح", "مكاسب", "تبرع", "تبرعات", "مساعدة", "حليف", "صعود", "انتعاش",
7
+ "ابتكار", "تطوير", "حسم", "ريادة", "امتياز", "سلام", "تعافي", "تسهيل", "دعم", "شراكة",
8
+ "تأهل", "ذهبية", "فضية", "برونزية", "ترقية", "جائزة", "تكريم", "مبادرة", "هدنة", "إيجابي",
9
+ "تقدم", "اعتماد", "موافقة", "تسهيلات", "شفاء", "علاج", "بشرى", "نجاحات", "تفاؤل", "فرحة",
10
+ "مكافأة", "تحسين", "توسيع", "إنقاذ", "صفقة", "عقد", "استثمار", "تسهيل", "إشادة", "ثراء",
11
+ "خير", "بركة", "بهجة", "سعادة", "انتصارات", "مكتسبات", "تكاتف", "تضامن", "مرونة", "تأهيل",
12
+ "إعفاء", "تسوية", "تسامح", "إنعاش", "بشائر", "طفرة", "تنمية", "إعادة إعمار", "جاهزية",
13
+
14
+ # English Positive Cues
15
+ "win", "winning", "winner", "victory", "triumph", "success", "successful", "growth", "grow",
16
+ "profit", "profitable", "boom", "booming", "recovery", "recover", "peace", "agreement",
17
+ "breakthrough", "milestone", "boost", "boosting", "gain", "gains", "innovation", "innovative",
18
+ "positive", "flourish", "thrive", "thriving", "achievement", "award", "awarded", "hero",
19
+ "heroic", "stability", "stable", "champion", "champions", "championship", "progress",
20
+ "partnership", "soar", "soaring", "upgrade", "advance", "advancement", "relief", "cure",
21
+ "prosper", "prosperity", "benefit", "beneficial", "glory", "glorious", "praise", "praised",
22
+ "accord", "reconciliation", "revival", "revive", "opportunity", "record", "surge", "surging",
23
+ "optimism", "optimistic", "solidarity", "resilience", "fund", "funded", "grant", "granted"
24
+ }
25
+
26
+ NEGATIVE_WORDS = {
27
+ # Arabic Negative Cues
28
+ "مقتل", "قتيل", "وفاة", "وفيات", "أزمة", "أزمات", "تراجع", "انخفاض", "خسارة", "خسائر",
29
+ "انهيار", "جريمة", "جرائم", "حادث", "حوادث", "كارثة", "كوارث", "تحذير", "عقوبات", "إفلاس",
30
+ "تهديد", "خيانة", "سرقة", "قلق", "زلزال", "فيضان", "حريق", "حرائق", "قتلى", "جرحى",
31
+ "مصابين", "إصابة", "حرب", "حروب", "قصف", "هجوم", "انفجار", "انفجارات", "توتر", "مأساة",
32
+ "ضحايا", "نزاع", "اعتقال", "اشتباكات", "شهداء", "عنف", "دمار", "تدمير", "مغادرة", "احتجاجات",
33
+ "نزوح", "لاجئين", "فقر", "مجاعة", "غلاء", "ركود", "إغلاق", "إلغاء", "إضراب", "اتهام",
34
+ "ادانة", "إدانة", "خراب", "سقوط", "نزيف", "تدهور", "مخاطر", "خطر", "فشل", "اضطراب",
35
+ "فضيحة", "فساد", "اختلاس", "احتيال", "مداهمة", "اعتقالات", "حصار", "احتلال", "قمع",
36
+
37
+ # English Negative Cues
38
+ "death", "dead", "died", "kill", "killed", "killing", "murder", "attack", "attacks", "war",
39
+ "warfare", "crisis", "crash", "crashed", "decline", "declining", "loss", "losses", "collapse",
40
+ "crime", "criminal", "warning", "warn", "threat", "threatening", "disaster", "disastrous",
41
+ "accident", "explosion", "explosions", "sanction", "sanctions", "inflation", "tragedy",
42
+ "tragic", "casualty", "casualties", "scandal", "fatal", "fatality", "strike", "strikes",
43
+ "conflict", "violence", "violent", "injury", "injured", "ruin", "ruined", "protest", "riot",
44
+ "poverty", "famine", "recession", "cancel", "cancelled", "fraud", "corruption", "arrest",
45
+ "arrested", "siege", "occupation", "devastation", "devastating", "slump", "downfall", "fear"
46
+ }
services/lexicons/stopwords_lexicon.py ADDED
@@ -0,0 +1,20 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Comprehensive Arabic & English Stopwords for NLP Tokenization & Tagging
2
+
3
+ STOP_WORDS = {
4
+ # Arabic Stopwords
5
+ "في", "من", "على", "عن", "أن", "إن", "إلى", "مع", "هذا", "هذه", "تم", "بعد", "قبل",
6
+ "كان", "كانت", "يكون", "تكون", "أو", "أم", "ثم", "حتى", "التي", "الذي", "الذين",
7
+ "اللذين", "اللتين", "بين", "حول", "ضد", "نحو", "كل", "جميع", "بعض", "غير", "ذات",
8
+ "ما", "ماذا", "منذ", "خلال", "عند", "أمام", "خلف", "تحت", "فوق", "كما", "لكن",
9
+ "غير", "أيضا", "فقط", "جدا", "قد", "لقد", "حيث", "إذا", "سوف", "سيكون", "اليوم",
10
+
11
+ # English Stopwords
12
+ "the", "a", "an", "in", "on", "at", "to", "for", "of", "with", "and", "is", "are",
13
+ "was", "were", "be", "been", "being", "have", "has", "had", "do", "does", "did",
14
+ "but", "or", "because", "as", "until", "while", "about", "against", "between",
15
+ "into", "through", "during", "before", "after", "above", "below", "from", "up",
16
+ "down", "in", "out", "over", "under", "again", "further", "then", "once", "here",
17
+ "there", "when", "where", "why", "how", "all", "any", "both", "each", "few", "more",
18
+ "most", "other", "some", "such", "no", "nor", "not", "only", "own", "same", "so",
19
+ "than", "too", "very", "s", "t", "can", "will", "just", "don", "should", "now"
20
+ }
services/local_nlp_classifier.py ADDED
@@ -0,0 +1,399 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Tier 2 Local NLP Classifier — Pre-trained Transformer Models.
3
+
4
+ This module provides near-LLM-quality enrichment using locally-downloaded
5
+ HuggingFace transformer models. It sits BETWEEN the HuggingFace API (Tier 1)
6
+ and the lexicon-based heuristics (Tier 3) in the fallback chain.
7
+
8
+ Models used:
9
+ - Category: facebook/bart-large-mnli (zero-shot classification, multilingual)
10
+ - Sentiment: CAMeL-Lab/bert-base-arabic-camelbert-msa-sentiment (Arabic)
11
+ distilbert-base-uncased-finetuned-sst-2-english (English)
12
+ - NER: CAMeL-Lab/bert-base-arabic-camelbert-msa-ner (Arabic)
13
+ dslim/bert-base-NER (English)
14
+ - Keywords: KeyBERT with paraphrase-multilingual-MiniLM-L12-v2
15
+ """
16
+
17
+ import re
18
+ import logging
19
+ from typing import Optional
20
+
21
+ logger = logging.getLogger(__name__)
22
+
23
+ # Lazy-loaded singletons — models are heavy, we only load them once and only when needed.
24
+ _category_pipeline = None
25
+ _sentiment_ar_pipeline = None
26
+ _sentiment_en_pipeline = None
27
+ _ner_ar_pipeline = None
28
+ _ner_en_pipeline = None
29
+ _keyword_model = None
30
+ _models_available = None # None = not checked yet, True/False after first attempt
31
+
32
+
33
+ # ---------------------------------------------------------------------------
34
+ # Our 13 categories (must match CATEGORIES_MAP keys in categories_lexicon.py)
35
+ # ---------------------------------------------------------------------------
36
+ CATEGORY_LABELS = [
37
+ "Sports", "Technology", "Gaming", "Business", "Health", "Science",
38
+ "Food", "Entertainment", "Politics", "Education", "Crime", "Weather", "Travel"
39
+ ]
40
+
41
+ # ---------------------------------------------------------------------------
42
+ # Region inference from NER-detected locations
43
+ # ---------------------------------------------------------------------------
44
+ from services.lexicons import MIDDLE_EAST_COUNTRIES, LOCATIONS_MAP
45
+
46
+
47
+ def _check_dependencies() -> bool:
48
+ """Check if transformers & torch are installed."""
49
+ global _models_available
50
+ if _models_available is not None:
51
+ return _models_available
52
+ try:
53
+ import transformers # noqa: F401
54
+ import torch # noqa: F401
55
+ _models_available = True
56
+ logger.info("NLP model dependencies (transformers, torch) are available.")
57
+ except ImportError as e:
58
+ _models_available = False
59
+ logger.warning(f"NLP model dependencies not installed ({e}). Tier 2 fallback disabled.")
60
+ return _models_available
61
+
62
+
63
+ # ========================= MODEL LOADERS =========================
64
+
65
+ def _get_category_pipeline():
66
+ """Zero-shot classification using BART-MNLI. Handles Arabic & English."""
67
+ global _category_pipeline
68
+ if _category_pipeline is not None:
69
+ return _category_pipeline
70
+ try:
71
+ from transformers import pipeline
72
+ _category_pipeline = pipeline(
73
+ "zero-shot-classification",
74
+ model="facebook/bart-large-mnli",
75
+ device=-1 # CPU only (GitHub Actions has no GPU)
76
+ )
77
+ logger.info("Loaded zero-shot category classifier (facebook/bart-large-mnli)")
78
+ except Exception as e:
79
+ logger.warning(f"Failed to load category classifier: {e}")
80
+ _category_pipeline = None
81
+ return _category_pipeline
82
+
83
+
84
+ def _get_sentiment_pipeline(language: str):
85
+ """Language-specific sentiment pipeline."""
86
+ global _sentiment_ar_pipeline, _sentiment_en_pipeline
87
+
88
+ if language == "ar":
89
+ if _sentiment_ar_pipeline is not None:
90
+ return _sentiment_ar_pipeline
91
+ try:
92
+ from transformers import pipeline
93
+ _sentiment_ar_pipeline = pipeline(
94
+ "text-classification",
95
+ model="CAMeL-Lab/bert-base-arabic-camelbert-msa-sentiment",
96
+ device=-1
97
+ )
98
+ logger.info("Loaded Arabic sentiment model (CAMeL-Lab/camelbert-msa-sentiment)")
99
+ except Exception as e:
100
+ logger.warning(f"Failed to load Arabic sentiment model: {e}")
101
+ _sentiment_ar_pipeline = None
102
+ return _sentiment_ar_pipeline
103
+ else:
104
+ if _sentiment_en_pipeline is not None:
105
+ return _sentiment_en_pipeline
106
+ try:
107
+ from transformers import pipeline
108
+ _sentiment_en_pipeline = pipeline(
109
+ "text-classification",
110
+ model="distilbert-base-uncased-finetuned-sst-2-english",
111
+ device=-1
112
+ )
113
+ logger.info("Loaded English sentiment model (distilbert-sst-2)")
114
+ except Exception as e:
115
+ logger.warning(f"Failed to load English sentiment model: {e}")
116
+ _sentiment_en_pipeline = None
117
+ return _sentiment_en_pipeline
118
+
119
+
120
+ def _get_ner_pipeline(language: str):
121
+ """Language-specific NER pipeline."""
122
+ global _ner_ar_pipeline, _ner_en_pipeline
123
+
124
+ if language == "ar":
125
+ if _ner_ar_pipeline is not None:
126
+ return _ner_ar_pipeline
127
+ try:
128
+ from transformers import pipeline
129
+ _ner_ar_pipeline = pipeline(
130
+ "ner",
131
+ model="CAMeL-Lab/bert-base-arabic-camelbert-msa-ner",
132
+ aggregation_strategy="simple",
133
+ device=-1
134
+ )
135
+ logger.info("Loaded Arabic NER model (CAMeL-Lab/camelbert-msa-ner)")
136
+ except Exception as e:
137
+ logger.warning(f"Failed to load Arabic NER model: {e}")
138
+ _ner_ar_pipeline = None
139
+ return _ner_ar_pipeline
140
+ else:
141
+ if _ner_en_pipeline is not None:
142
+ return _ner_en_pipeline
143
+ try:
144
+ from transformers import pipeline
145
+ _ner_en_pipeline = pipeline(
146
+ "ner",
147
+ model="dslim/bert-base-NER",
148
+ aggregation_strategy="simple",
149
+ device=-1
150
+ )
151
+ logger.info("Loaded English NER model (dslim/bert-base-NER)")
152
+ except Exception as e:
153
+ logger.warning(f"Failed to load English NER model: {e}")
154
+ _ner_en_pipeline = None
155
+ return _ner_en_pipeline
156
+
157
+
158
+ def _get_keyword_model():
159
+ """KeyBERT with multilingual sentence-transformer embeddings."""
160
+ global _keyword_model
161
+ if _keyword_model is not None:
162
+ return _keyword_model
163
+ try:
164
+ from keybert import KeyBERT
165
+ _keyword_model = KeyBERT(model="paraphrase-multilingual-MiniLM-L12-v2")
166
+ logger.info("Loaded KeyBERT keyword extractor (multilingual-MiniLM)")
167
+ except Exception as e:
168
+ logger.warning(f"Failed to load KeyBERT: {e}")
169
+ _keyword_model = None
170
+ return _keyword_model
171
+
172
+
173
+ # ========================= ENRICHMENT =========================
174
+
175
+ class LocalNLPClassifier:
176
+ """
177
+ Tier 2 NLP enrichment engine using pre-trained transformer models.
178
+
179
+ Usage:
180
+ classifier = LocalNLPClassifier()
181
+ if classifier.is_available():
182
+ enriched_article = classifier.enrich_article(article)
183
+ """
184
+
185
+ def is_available(self) -> bool:
186
+ """Check if the required ML libraries are installed."""
187
+ return _check_dependencies()
188
+
189
+ def enrich_article(self, article) -> "Article":
190
+ """
191
+ Enrich a single article using local transformer models.
192
+ Each sub-task is wrapped in try/except so a failure in one
193
+ (e.g. sentiment) does not block the others (e.g. NER).
194
+ """
195
+ text = f"{article.title} {article.summary or ''}"
196
+ lang = getattr(article, 'language', 'en')
197
+
198
+ # 1. Category (zero-shot classification)
199
+ try:
200
+ self._classify_category(article, text)
201
+ except Exception as e:
202
+ logger.warning(f"Tier 2 category classification failed: {e}")
203
+
204
+ # 2. Sentiment
205
+ try:
206
+ self._classify_sentiment(article, text, lang)
207
+ except Exception as e:
208
+ logger.warning(f"Tier 2 sentiment analysis failed: {e}")
209
+
210
+ # 3. Named Entity Recognition -> people, organizations, locations, country, region
211
+ try:
212
+ self._extract_entities(article, text, lang)
213
+ except Exception as e:
214
+ logger.warning(f"Tier 2 NER failed: {e}")
215
+
216
+ # 4. Keywords & Tags
217
+ try:
218
+ self._extract_keywords(article, text)
219
+ except Exception as e:
220
+ logger.warning(f"Tier 2 keyword extraction failed: {e}")
221
+
222
+ # 5. Importance scoring (derived from NER entity density + category)
223
+ try:
224
+ self._score_importance(article, text)
225
+ except Exception as e:
226
+ logger.warning(f"Tier 2 importance scoring failed: {e}")
227
+
228
+ # 6. Reading time estimation
229
+ if not article.readingTime:
230
+ article.readingTime = self._estimate_reading_time(text)
231
+
232
+ return article
233
+
234
+ # ---------------------------------------------------------------
235
+ # Sub-task implementations
236
+ # ---------------------------------------------------------------
237
+
238
+ def _classify_category(self, article, text: str):
239
+ """Zero-shot classification into our 13 news categories."""
240
+ if article.category and article.category not in ("General", "Middle East", "Egypt", "World", "North Africa"):
241
+ return # Already classified by the RSS source metadata
242
+
243
+ pipe = _get_category_pipeline()
244
+ if pipe is None:
245
+ return
246
+
247
+ # Truncate to 512 tokens for BART
248
+ result = pipe(text[:1024], candidate_labels=CATEGORY_LABELS, multi_label=False)
249
+ if result and result['scores'][0] > 0.25:
250
+ article.category = result['labels'][0]
251
+ # Subcategory: use the second-highest label if confident
252
+ if len(result['labels']) > 1 and result['scores'][1] > 0.20:
253
+ article.subcategory = result['labels'][1]
254
+ else:
255
+ article.category = article.category or "General"
256
+
257
+ def _classify_sentiment(self, article, text: str, lang: str):
258
+ """Sentiment analysis using language-specific pre-trained models."""
259
+ pipe = _get_sentiment_pipeline(lang)
260
+ if pipe is None:
261
+ return
262
+
263
+ result = pipe(text[:512])
264
+ if not result:
265
+ return
266
+
267
+ label = result[0]['label'].upper()
268
+ score = result[0]['score']
269
+
270
+ if lang == "ar":
271
+ # CAMeLBERT sentiment outputs: positive, negative, neutral
272
+ label_map = {"POSITIVE": "Positive", "NEGATIVE": "Negative", "NEUTRAL": "Neutral"}
273
+ article.sentiment = label_map.get(label, "Neutral")
274
+ else:
275
+ # DistilBERT SST-2 outputs: POSITIVE, NEGATIVE
276
+ if label == "POSITIVE":
277
+ article.sentiment = "Positive"
278
+ elif label == "NEGATIVE":
279
+ article.sentiment = "Negative"
280
+ else:
281
+ article.sentiment = "Neutral"
282
+
283
+ # Mixed sentiment: if confidence is low (close to 50/50)
284
+ if 0.45 < score < 0.60:
285
+ article.sentiment = "Mixed"
286
+
287
+ def _extract_entities(self, article, text: str, lang: str):
288
+ """NER extraction: people, organizations, locations."""
289
+ pipe = _get_ner_pipeline(lang)
290
+ if pipe is None:
291
+ return
292
+
293
+ entities = pipe(text[:512])
294
+ if not entities:
295
+ return
296
+
297
+ people = []
298
+ organizations = []
299
+ locations = []
300
+
301
+ for ent in entities:
302
+ entity_group = ent.get('entity_group', '')
303
+ word = ent.get('word', '').strip()
304
+ score = ent.get('score', 0)
305
+
306
+ # Filter low-confidence entities
307
+ if score < 0.5 or len(word) < 2:
308
+ continue
309
+
310
+ # Clean up subword tokens (##prefix)
311
+ word = word.replace('##', '')
312
+
313
+ if entity_group in ('PER', 'B-PER', 'I-PER'):
314
+ if word not in people:
315
+ people.append(word)
316
+ elif entity_group in ('ORG', 'B-ORG', 'I-ORG'):
317
+ if word not in organizations:
318
+ organizations.append(word)
319
+ elif entity_group in ('LOC', 'B-LOC', 'I-LOC', 'GPE', 'B-GPE', 'I-GPE'):
320
+ if word not in locations:
321
+ locations.append(word)
322
+
323
+ article.people = people[:10] # Cap at 10
324
+ article.organizations = organizations[:10]
325
+ article.locations = locations[:10] if locations else article.locations
326
+
327
+ # Derive country and region from detected locations
328
+ if locations:
329
+ # Try to map NER location names to our gazetteer
330
+ for loc in locations:
331
+ loc_lower = loc.lower()
332
+ if loc_lower in LOCATIONS_MAP:
333
+ article.country = LOCATIONS_MAP[loc_lower]
334
+ article.region = "Middle East" if article.country in MIDDLE_EAST_COUNTRIES else "Global"
335
+ break
336
+ if not article.country:
337
+ article.country = locations[0]
338
+ article.region = "Global"
339
+
340
+ def _extract_keywords(self, article, text: str):
341
+ """KeyBERT keyword extraction with multilingual embeddings."""
342
+ kw_model = _get_keyword_model()
343
+ if kw_model is None:
344
+ return
345
+
346
+ try:
347
+ keywords = kw_model.extract_keywords(
348
+ text,
349
+ keyphrase_ngram_range=(1, 2),
350
+ stop_words=None, # KeyBERT handles multilingual stop words
351
+ top_n=12,
352
+ use_maxsum=True,
353
+ nr_candidates=20
354
+ )
355
+ except Exception:
356
+ return
357
+
358
+ if keywords:
359
+ article.keywords = [kw[0] for kw in keywords[:6]]
360
+ article.tags = [kw[0] for kw in keywords[:10]]
361
+
362
+ def _score_importance(self, article, text: str):
363
+ """
364
+ Importance scoring (1-10) based on:
365
+ - Number of named entities (more entities = more important)
366
+ - High-impact trigger words
367
+ - Category weight
368
+ """
369
+ score = 5 # Base importance
370
+
371
+ # Named entity density bonus
372
+ entity_count = len(article.people or []) + len(article.organizations or []) + len(article.locations or [])
373
+ if entity_count >= 5:
374
+ score += 2
375
+ elif entity_count >= 2:
376
+ score += 1
377
+
378
+ # High-impact trigger words
379
+ high_impact_terms = [
380
+ "عاجل", "قمة", "حرب", "وفاة", "رئيس", "كوارث", "اغتيال", "انقلاب", "زلزال",
381
+ "breaking", "war", "president", "disaster", "crisis", "urgent", "killed",
382
+ "earthquake", "assassination", "coup", "summit", "death"
383
+ ]
384
+ text_lower = text.lower()
385
+ if any(term in text_lower for term in high_impact_terms):
386
+ score += 3
387
+
388
+ # Category bonus
389
+ high_importance_categories = {"Politics", "Crime", "Health", "Science"}
390
+ if article.category in high_importance_categories:
391
+ score += 1
392
+
393
+ article.importance = min(10, max(1, score))
394
+
395
+ def _estimate_reading_time(self, text: str) -> int:
396
+ """Estimate reading time in minutes from word count."""
397
+ clean_text = re.sub(r'<[^>]+>', '', text)
398
+ words = len(clean_text.split())
399
+ return max(1, round(words / 180))