Bjornpool commited on
Commit
95f9219
·
0 Parent(s):

feat: initial HF Spaces deployment

Browse files
.gitignore ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ .env
2
+ __pycache__/
3
+ *.pyc
4
+ *.pyo
5
+ saved_articles.json
6
+ benchmark_results/
7
+ debug_adapters.py
8
+ truthscan_report.*
9
+ truthscan_test.py
10
+ setup_test_env.bat
Dockerfile ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ FROM python:3.11-slim
2
+ WORKDIR /app
3
+ COPY requirements.txt .
4
+ RUN pip install --no-cache-dir -r requirements.txt
5
+ COPY . .
6
+ EXPOSE 7860
7
+ CMD ["uvicorn", "app.main:app", "--host", "0.0.0.0", "--port", "7860"]
app/__init__.py ADDED
File without changes
app/benchmark.py ADDED
@@ -0,0 +1,240 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Moduł benchmarkowania modeli NLP.
3
+
4
+ Mierzy czas inferencji, rozkład sentymentów i prawdopodobieństwo fake news
5
+ dla każdego adaptera (roberta, xlm-roberta, norbert) na próbce tekstów
6
+ z trzech grup językowych (en, pl, no).
7
+
8
+ Użycie standalone:
9
+ python -m app.benchmark
10
+
11
+ Użycie z API:
12
+ GET /benchmark
13
+ GET /benchmark?adapters=roberta,xlm-roberta&langs=en,pl
14
+ """
15
+
16
+ import csv
17
+ import json
18
+ import time
19
+ from collections import Counter
20
+ from pathlib import Path
21
+ from typing import Dict, List, Optional
22
+
23
+ from .nlp_service import get_adapter, ModelAdapter, analyze_news
24
+
25
+ # ---------------------------------------------------------------------------
26
+ # Próbka tekstów testowych
27
+ # ---------------------------------------------------------------------------
28
+
29
+ SAMPLE_TEXTS: Dict[str, List[str]] = {
30
+ "en": [
31
+ "The government announced new economic reforms to boost growth and reduce unemployment.",
32
+ "Flooding devastated coastal towns overnight, leaving thousands homeless.",
33
+ "Scientists discover a new vaccine that shows 95% efficacy against the virus.",
34
+ "Stock markets surged to record highs after positive inflation data.",
35
+ "A major scandal erupted as leaked documents exposed corporate corruption.",
36
+ "The peace talks collapsed after both sides failed to reach an agreement.",
37
+ "Renewable energy investments hit an all-time high this quarter.",
38
+ "Crime rates in the capital have dropped significantly over the past year.",
39
+ ],
40
+ "pl": [
41
+ "Rząd ogłosił nowe reformy gospodarcze mające na celu pobudzenie wzrostu.",
42
+ "Powódź zniszczyła nadmorskie miejscowości, tysiące osób zostało bez dachu.",
43
+ "Naukowcy odkryli szczepionkę o 95-procentowej skuteczności przeciw wirusowi.",
44
+ "Giełda osiągnęła rekordowe poziomy po pozytywnych danych o inflacji.",
45
+ "Wybuchł wielki skandal po ujawnieniu dokumentów o korupcji korporacyjnej.",
46
+ "Rozmowy pokojowe załamały się po niepowodzeniu negocjacji.",
47
+ "Inwestycje w energię odnawialną osiągnęły historyczny rekord w tym kwartale.",
48
+ "Wskaźniki przestępczości w stolicy znacząco spadły w ciągu ostatniego roku.",
49
+ ],
50
+ "no": [
51
+ "Regjeringen kunngjorde nye økonomiske reformer for å øke veksten.",
52
+ "Flom ødela kystbyer over natten og etterlot tusenvis uten hjem.",
53
+ "Forskere oppdaget en vaksine med 95 prosent effektivitet mot viruset.",
54
+ "Aksjemarkedene steg til rekordhøyder etter positive inflasjonsdata.",
55
+ "En stor skandale brøt ut da lekkede dokumenter avslørte korrupsjon.",
56
+ "Fredssamtalene brøt sammen etter at begge sider ikke klarte å bli enige.",
57
+ "Investeringer i fornybar energi nådde en historisk topp dette kvartalet.",
58
+ "Kriminalitetsratene i hovedstaden har falt betydelig det siste året.",
59
+ ],
60
+ }
61
+
62
+ # ---------------------------------------------------------------------------
63
+ # Typy wyników
64
+ # ---------------------------------------------------------------------------
65
+
66
+ BenchmarkResult = Dict # TypedDict zastąpiony zwykłym Dict dla czytelności
67
+
68
+ # ---------------------------------------------------------------------------
69
+ # Funkcje benchmarkowania
70
+ # ---------------------------------------------------------------------------
71
+
72
+ def _run_single(
73
+ adapter: ModelAdapter,
74
+ text: str,
75
+ lang: str,
76
+ ) -> Dict:
77
+ """Uruchamia analyze_news dla jednego tekstu i mierzy czas."""
78
+ start = time.perf_counter()
79
+ result = analyze_news(text, lang=lang, adapter=adapter)
80
+ elapsed_ms = (time.perf_counter() - start) * 1000
81
+ return {**result, "inference_time_ms": elapsed_ms}
82
+
83
+
84
+ def run_benchmark(
85
+ adapter_names: Optional[List[str]] = None,
86
+ langs: Optional[List[str]] = None,
87
+ ) -> List[BenchmarkResult]:
88
+ """
89
+ Uruchamia benchmark dla wskazanych adapterów i języków.
90
+
91
+ Args:
92
+ adapter_names: Lista nazw adapterów; None = wszystkie trzy.
93
+ langs: Lista kodów języków; None = ['en', 'pl', 'no'].
94
+
95
+ Returns:
96
+ Lista słowników z wynikami — jeden wpis na kombinację adapter × język.
97
+ """
98
+ if adapter_names is None:
99
+ adapter_names = ["roberta", "xlm-roberta", "norbert"]
100
+ if langs is None:
101
+ langs = ["en", "pl", "no"]
102
+
103
+ results: List[BenchmarkResult] = []
104
+
105
+ for adapter_name in adapter_names:
106
+ try:
107
+ adapter = get_adapter(adapter_name)
108
+ except ValueError as exc:
109
+ results.append({
110
+ "adapter_name": adapter_name,
111
+ "error": str(exc),
112
+ })
113
+ continue
114
+
115
+ for lang in langs:
116
+ texts = SAMPLE_TEXTS.get(lang, [])
117
+ if not texts:
118
+ continue
119
+
120
+ per_text: List[Dict] = []
121
+ for text in texts:
122
+ try:
123
+ per_text.append(_run_single(adapter, text, lang))
124
+ except Exception as exc:
125
+ per_text.append({
126
+ "sentiment": None,
127
+ "fake_probability": None,
128
+ "sentiment_score": None,
129
+ "inference_time_ms": None,
130
+ "error": str(exc),
131
+ })
132
+
133
+ # Agregacja
134
+ valid = [r for r in per_text if r.get("inference_time_ms") is not None]
135
+ times = [r["inference_time_ms"] for r in valid]
136
+ fakes = [r["fake_probability"] for r in valid if r.get("fake_probability") is not None]
137
+ sentiments = [r["sentiment"] for r in valid if r.get("sentiment")]
138
+
139
+ results.append({
140
+ "adapter_name": adapter_name,
141
+ "language": lang,
142
+ "sample_size": len(texts),
143
+ "successful_runs": len(valid),
144
+ "avg_inference_time_ms": round(sum(times) / len(times), 2) if times else None,
145
+ "min_inference_time_ms": round(min(times), 2) if times else None,
146
+ "max_inference_time_ms": round(max(times), 2) if times else None,
147
+ "avg_fake_probability": round(sum(fakes) / len(fakes), 2) if fakes else None,
148
+ "sentiments_distribution": dict(Counter(sentiments)),
149
+ "per_text": per_text,
150
+ })
151
+
152
+ return results
153
+
154
+
155
+ # ---------------------------------------------------------------------------
156
+ # Eksport wyników
157
+ # ---------------------------------------------------------------------------
158
+
159
+ def export_json(results: List[BenchmarkResult], path: Path) -> None:
160
+ """Zapisuje pełne wyniki (z per_text) do pliku JSON."""
161
+ path.parent.mkdir(parents=True, exist_ok=True)
162
+ with open(path, "w", encoding="utf-8") as fh:
163
+ json.dump(results, fh, ensure_ascii=False, indent=2)
164
+
165
+
166
+ def export_csv(results: List[BenchmarkResult], path: Path) -> None:
167
+ """
168
+ Zapisuje wyniki zbiorcze (bez per_text) do pliku CSV.
169
+ Jeden wiersz = jedna kombinacja adapter × język.
170
+ """
171
+ path.parent.mkdir(parents=True, exist_ok=True)
172
+ summary_fields = [
173
+ "adapter_name", "language", "sample_size", "successful_runs",
174
+ "avg_inference_time_ms", "min_inference_time_ms", "max_inference_time_ms",
175
+ "avg_fake_probability", "sentiments_distribution",
176
+ ]
177
+ with open(path, "w", newline="", encoding="utf-8") as fh:
178
+ writer = csv.DictWriter(fh, fieldnames=summary_fields, extrasaction="ignore")
179
+ writer.writeheader()
180
+ for row in results:
181
+ if "error" in row:
182
+ continue
183
+ flat = {k: row.get(k) for k in summary_fields}
184
+ # Rozkład sentymentów jako string JSON w komórce CSV
185
+ flat["sentiments_distribution"] = json.dumps(
186
+ row.get("sentiments_distribution", {}), ensure_ascii=False
187
+ )
188
+ writer.writerow(flat)
189
+
190
+
191
+ def _summary_only(results: List[BenchmarkResult]) -> List[BenchmarkResult]:
192
+ """Zwraca wyniki bez pola per_text (lżejsza odpowiedź HTTP)."""
193
+ return [{k: v for k, v in r.items() if k != "per_text"} for r in results]
194
+
195
+
196
+ # ---------------------------------------------------------------------------
197
+ # Uruchomienie standalone
198
+ # ---------------------------------------------------------------------------
199
+
200
+ if __name__ == "__main__":
201
+ import argparse
202
+
203
+ parser = argparse.ArgumentParser(description="TruthScan NLP benchmark")
204
+ parser.add_argument(
205
+ "--adapters", default="roberta,xlm-roberta,norbert",
206
+ help="Przecinkowa lista adapterów (domyślnie: wszystkie)",
207
+ )
208
+ parser.add_argument(
209
+ "--langs", default="en,pl,no",
210
+ help="Przecinkowa lista języków (domyślnie: en,pl,no)",
211
+ )
212
+ parser.add_argument(
213
+ "--out-dir", default="benchmark_results",
214
+ help="Katalog wyjściowy dla plików JSON i CSV",
215
+ )
216
+ args = parser.parse_args()
217
+
218
+ adapter_names = [a.strip() for a in args.adapters.split(",")]
219
+ langs = [l.strip() for l in args.langs.split(",")]
220
+ out_dir = Path(args.out_dir)
221
+
222
+ print(f"Uruchamiam benchmark: adaptery={adapter_names}, języki={langs}")
223
+ results = run_benchmark(adapter_names=adapter_names, langs=langs)
224
+
225
+ json_path = out_dir / "benchmark.json"
226
+ csv_path = out_dir / "benchmark.csv"
227
+ export_json(results, json_path)
228
+ export_csv(results, csv_path)
229
+
230
+ print(f"Wyniki zapisane: {json_path}, {csv_path}")
231
+ for r in _summary_only(results):
232
+ if "error" in r:
233
+ print(f" [{r['adapter_name']}] BŁĄD: {r['error']}")
234
+ else:
235
+ print(
236
+ f" [{r['adapter_name']:12s} / {r['language']}] "
237
+ f"avg={r['avg_inference_time_ms']} ms "
238
+ f"fake={r['avg_fake_probability']}% "
239
+ f"sentiments={r['sentiments_distribution']}"
240
+ )
app/config.py ADDED
@@ -0,0 +1,61 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Centralna konfiguracja aplikacji oraz stałe wykorzystywane w wielu modułach.
3
+ """
4
+
5
+ import os
6
+ from pathlib import Path
7
+
8
+ try:
9
+ from dotenv import load_dotenv
10
+ load_dotenv(Path(__file__).resolve().parent.parent / ".env")
11
+ except ImportError:
12
+ pass # python-dotenv opcjonalne; zmienne można też ustawić w środowisku
13
+
14
+ # Konfiguracja CORS
15
+ # Uwaga: credentials=True jest niekompatybilne z origins=["*"] w przeglądarkach
16
+ # (spec CORS odrzuca tę kombinację). Dla dev/thesis używamy credentials=False.
17
+ CORS_ALLOW_ORIGINS = ["*"]
18
+ CORS_ALLOW_CREDENTIALS = False
19
+ CORS_ALLOW_METHODS = ["GET", "POST", "DELETE", "OPTIONS"]
20
+ CORS_ALLOW_HEADERS = ["*"]
21
+
22
+ # Lista obsługiwanych źródeł RSS
23
+ NEWS_FEEDS = {
24
+ # Angielskie
25
+ "BBC": "https://feeds.bbci.co.uk/news/rss.xml",
26
+ "CNN": "http://rss.cnn.com/rss/edition.rss",
27
+ "NYTimes": "https://rss.nytimes.com/services/xml/rss/nyt/HomePage.xml",
28
+ "Guardian": "https://www.theguardian.com/world/rss",
29
+ "AlJazeera": "https://www.aljazeera.com/xml/rss/all.xml",
30
+ # Polskie
31
+ "Money": "https://www.money.pl/rss/",
32
+ "PolsatNews": "https://www.polsatnews.pl/rss/wszystkie.xml",
33
+ "TVN24": "https://tvn24.pl/najnowsze.xml",
34
+ "SpidersWeb": "https://spidersweb.pl/feed",
35
+ "Bankier": "https://www.bankier.pl/rss/wiadomosci.xml",
36
+ # Norweskie
37
+ "NRK": "https://www.nrk.no/toppsaker.rss",
38
+ "VG": "https://www.vg.no/rss/feed/?limit=10",
39
+ "E24": "https://e24.no/rss",
40
+ "Aftenposten": "https://www.aftenposten.no/rss",
41
+ }
42
+
43
+ # Mapowanie wyników analizy sentymentu na etykiety językowe
44
+ SENTIMENT_MAP = {
45
+ "negative": {"pl": "Negatywne", "en": "Negative", "no": "Negativt"},
46
+ "neutral": {"pl": "Neutralne", "en": "Neutral", "no": "Nøytralt"},
47
+ "positive": {"pl": "Pozytywne", "en": "Positive", "no": "Positivt"},
48
+ }
49
+
50
+ # Ścieżka do pliku z zapisanymi artykułami
51
+ SAVED_FILE = Path("saved_articles.json")
52
+
53
+ # Czas życia cache (sekundy)
54
+ CACHE_TTL_SECONDS = 120
55
+
56
+ # Konfiguracja Redis (jeśli używany jako backend cache)
57
+ REDIS_URL = os.getenv("REDIS_URL", "redis://localhost:6379")
58
+ CACHE_TTL = 300
59
+
60
+ # Klucz API Anthropic (Claude)
61
+ ANTHROPIC_API_KEY = os.getenv("ANTHROPIC_API_KEY")
app/main.py ADDED
@@ -0,0 +1,55 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Główna konfiguracja i uruchomienie aplikacji FastAPI.
3
+ """
4
+
5
+ from fastapi import FastAPI, Request
6
+ from fastapi.responses import Response
7
+ from fastapi_cache import FastAPICache
8
+ from fastapi_cache.backends.inmemory import InMemoryBackend
9
+
10
+ from .routes import misc, news, saved, compare
11
+
12
+ app = FastAPI()
13
+
14
+ _CORS_HEADERS = {
15
+ "Access-Control-Allow-Origin": "*",
16
+ "Access-Control-Allow-Methods": "GET, POST, DELETE, OPTIONS",
17
+ "Access-Control-Allow-Headers": "*",
18
+ "Access-Control-Max-Age": "86400",
19
+ }
20
+
21
+
22
+ @app.middleware("http")
23
+ async def add_cors_headers(request: Request, call_next):
24
+ # Odpowiedź na preflight OPTIONS bez przekazywania do routera
25
+ if request.method == "OPTIONS":
26
+ return Response(status_code=204, headers=_CORS_HEADERS)
27
+
28
+ response = await call_next(request)
29
+
30
+ for key, value in _CORS_HEADERS.items():
31
+ response.headers[key] = value
32
+
33
+ return response
34
+
35
+
36
+ @app.on_event("startup")
37
+ async def startup():
38
+ FastAPICache.init(InMemoryBackend(), prefix="news-cache")
39
+ print("✓ Cache initialized (5 minut TTL)")
40
+
41
+
42
+ app.include_router(misc.router)
43
+ app.include_router(news.router)
44
+ app.include_router(saved.router)
45
+ app.include_router(compare.router)
46
+
47
+
48
+ @app.get("/")
49
+ async def root():
50
+ return {"message": "TruthScan API", "status": "running", "cached": True}
51
+
52
+
53
+ @app.get("/health")
54
+ async def health_check():
55
+ return {"status": "healthy"}
app/models.py ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Modele danych wykorzystywane do walidacji i serializacji artykułów.
3
+ """
4
+
5
+ from pydantic import BaseModel
6
+
7
+
8
+ class Article(BaseModel):
9
+ # Model artykułu wykorzystywany w komunikacji API
10
+ title: str
11
+ link: str
12
+ summary: str
13
+ published: str
14
+ sentiment: str
15
+ fake_probability: float
16
+ source: str
app/nlp_service.py ADDED
@@ -0,0 +1,433 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Serwis NLP z architekturą plug-in.
3
+
4
+ Każdy model jest reprezentowany przez adapter dziedziczący z ModelAdapter.
5
+ Publiczne API (analyze_news, analyze_news_batch) pozostaje niezmienione,
6
+ więc routes/news.py nie wymaga modyfikacji.
7
+
8
+ Fake news detection używa jednej globalnej instancji BART (_shared_bart)
9
+ współdzielonej przez wszystkie adaptery — model ładowany jest leniwie
10
+ przy pierwszym wywołaniu get_shared_bart().
11
+ """
12
+
13
+ from abc import ABC, abstractmethod
14
+ from typing import Dict, Any, List, Optional
15
+ from concurrent.futures import ThreadPoolExecutor
16
+
17
+ from transformers import pipeline
18
+
19
+ from .config import SENTIMENT_MAP
20
+
21
+
22
+ # ---------------------------------------------------------------------------
23
+ # Współdzielony BART do wykrywania fake news (jeden egzemplarz dla wszystkich)
24
+ # ---------------------------------------------------------------------------
25
+
26
+ _shared_bart = None
27
+
28
+
29
+ def get_shared_bart():
30
+ """
31
+ Zwraca globalną instancję potoku zero-shot-classification (BART).
32
+ Ładuje model przy pierwszym wywołaniu (lazy init).
33
+ """
34
+ global _shared_bart
35
+ if _shared_bart is None:
36
+ _shared_bart = pipeline(
37
+ "zero-shot-classification",
38
+ model="facebook/bart-large-mnli",
39
+ )
40
+ return _shared_bart
41
+
42
+
43
+ def analyze_fake_news_shared(text: str) -> float:
44
+ """
45
+ Klasyfikuje tekst jako real/fake używając współdzielonego BART.
46
+
47
+ Zwraca:
48
+ Prawdopodobieństwo fake news w procentach (0.0 – 100.0).
49
+ """
50
+ result = get_shared_bart()(text, candidate_labels=["real", "fake"])
51
+ for lbl, score in zip(result["labels"], result["scores"]):
52
+ if lbl == "fake":
53
+ return round(score * 100, 2)
54
+ return 0.0
55
+
56
+
57
+ # ---------------------------------------------------------------------------
58
+ # Klasa bazowa
59
+ # ---------------------------------------------------------------------------
60
+
61
+ class ModelAdapter(ABC):
62
+ """
63
+ Abstrakcyjny adapter modelu NLP.
64
+
65
+ Każdy konkretny adapter implementuje:
66
+ - analyze_sentiment(text) -> {"label": str, "score": float}
67
+ - analyze_fake_news(text) -> {"labels": List[str], "scores": List[float]}
68
+
69
+ analyze_fake_news deleguje do współdzielonego BART (get_shared_bart),
70
+ więc każda podklasa może użyć domyślnej implementacji z tej klasy bazowej.
71
+ """
72
+
73
+ @property
74
+ @abstractmethod
75
+ def name(self) -> str:
76
+ ...
77
+
78
+ @property
79
+ @abstractmethod
80
+ def supported_languages(self) -> List[str]:
81
+ ...
82
+
83
+ @abstractmethod
84
+ def analyze_sentiment(self, text: str) -> Dict[str, Any]:
85
+ """
86
+ Zwraca {"label": str, "score": float}
87
+ gdzie label to: 'positive' | 'negative' | 'neutral'
88
+ """
89
+ ...
90
+
91
+ def analyze_fake_news(self, text: str) -> Dict[str, Any]:
92
+ """
93
+ Klasyfikuje tekst używając współdzielonego BART.
94
+ Podklasy mogą nadpisać, ale domyślnie korzystają z _shared_bart.
95
+ """
96
+ result = get_shared_bart()(text, candidate_labels=["real", "fake"])
97
+ return {"labels": result["labels"], "scores": result["scores"]}
98
+
99
+
100
+ # ---------------------------------------------------------------------------
101
+ # Helpers
102
+ # ---------------------------------------------------------------------------
103
+
104
+ def _extract_pipeline_result(adapter_name: str, raw: Any) -> Dict[str, Any]:
105
+ """
106
+ Bezpiecznie wyciąga słownik {label, score} z surowego outputu pipeline.
107
+
108
+ Pipeline text-classification może zwrócić:
109
+ - [{"label": ..., "score": ...}] (return_all_scores=False)
110
+ - [[{"label": ..., "score": ...}, ...]] (return_all_scores=True / top_k=None)
111
+ """
112
+ print(f"[DEBUG] {adapter_name}: raw = {raw!r}")
113
+ try:
114
+ item = raw[0]
115
+ except (IndexError, TypeError) as exc:
116
+ print(f"[DEBUG] {adapter_name}: raw[0] failed: {exc}")
117
+ return {"label": "", "score": 0.0}
118
+
119
+ if isinstance(item, dict):
120
+ result = item
121
+ elif isinstance(item, list):
122
+ # return_all_scores=True zwraca listę list — bierzemy element o najwyższym score
123
+ if not item:
124
+ print(f"[DEBUG] {adapter_name}: item jest pustą listą")
125
+ return {"label": "", "score": 0.0}
126
+ result = max(item, key=lambda x: x.get("score", 0.0))
127
+ else:
128
+ print(f"[DEBUG] {adapter_name}: nieoczekiwany typ item: {type(item)!r}, raw={raw!r}")
129
+ return {"label": "", "score": 0.0}
130
+
131
+ print(f"[DEBUG] {adapter_name}: extracted = {result!r}")
132
+ return result
133
+
134
+
135
+ def _normalize_sentiment_label(raw_label: str) -> str:
136
+ label = (raw_label or "").strip().lower()
137
+
138
+ _POSITIVE = {"positive", "pos", "label_2", "2", "very positive"}
139
+ _NEGATIVE = {"negative", "neg", "label_0", "0", "very negative"}
140
+ # "mixed" (NorBERT3 sentence-sentiment) → traktujemy jako neutral
141
+ _NEUTRAL = {"neutral", "mixed", "label_1"}
142
+
143
+ if label in _POSITIVE or label.startswith("pos"):
144
+ return "positive"
145
+ if label in _NEGATIVE or label.startswith("neg"):
146
+ return "negative"
147
+ if label in _NEUTRAL:
148
+ return "neutral"
149
+ return "neutral"
150
+
151
+
152
+ # ---------------------------------------------------------------------------
153
+ # Adaptery — każdy ładuje tylko swój model sentymentu
154
+ # ---------------------------------------------------------------------------
155
+
156
+ class RoBERTaAdapter(ModelAdapter):
157
+ """
158
+ Adapter anglojęzyczny.
159
+ sentyment: cardiffnlp/twitter-roberta-base-sentiment-latest
160
+ fake news: współdzielony BART (ModelAdapter.analyze_fake_news)
161
+ """
162
+
163
+ def __init__(self) -> None:
164
+ self._sentiment_pipe = None
165
+
166
+ def _load(self) -> None:
167
+ if self._sentiment_pipe is None:
168
+ self._sentiment_pipe = pipeline(
169
+ "text-classification",
170
+ model="cardiffnlp/twitter-roberta-base-sentiment-latest",
171
+ return_all_scores=False,
172
+ )
173
+
174
+ @property
175
+ def name(self) -> str:
176
+ return "roberta"
177
+
178
+ @property
179
+ def supported_languages(self) -> List[str]:
180
+ return ["en"]
181
+
182
+ def analyze_sentiment(self, text: str) -> Dict[str, Any]:
183
+ self._load()
184
+ raw = self._sentiment_pipe(text)
185
+ result = _extract_pipeline_result(self.name, raw)
186
+ return {
187
+ "label": _normalize_sentiment_label(result.get("label", "")),
188
+ "score": float(result.get("score", 0.0)),
189
+ }
190
+
191
+
192
+ class XLMRoBERTaAdapter(ModelAdapter):
193
+ """
194
+ Adapter wielojęzyczny (pl, en, no).
195
+ sentyment: cardiffnlp/twitter-xlm-roberta-base-sentiment
196
+ fake news: współdzielony BART
197
+ """
198
+
199
+ def __init__(self) -> None:
200
+ self._sentiment_pipe = None
201
+
202
+ def _load(self) -> None:
203
+ if self._sentiment_pipe is None:
204
+ self._sentiment_pipe = pipeline(
205
+ "text-classification",
206
+ model="cardiffnlp/twitter-xlm-roberta-base-sentiment",
207
+ return_all_scores=False,
208
+ )
209
+
210
+ @property
211
+ def name(self) -> str:
212
+ return "xlm-roberta"
213
+
214
+ @property
215
+ def supported_languages(self) -> List[str]:
216
+ return ["pl", "en", "no"]
217
+
218
+ def analyze_sentiment(self, text: str) -> Dict[str, Any]:
219
+ self._load()
220
+ raw = self._sentiment_pipe(text)
221
+ result = _extract_pipeline_result(self.name, raw)
222
+ return {
223
+ "label": _normalize_sentiment_label(result.get("label", "")),
224
+ "score": float(result.get("score", 0.0)),
225
+ }
226
+
227
+
228
+ class HerBERTAdapter(ModelAdapter):
229
+ """
230
+ Adapter dla języka polskiego (HerBERT).
231
+ sentyment: allegro/herbert-base-cased (model bazowy — wymaga fine-tuningu)
232
+ fake news: współdzielony BART
233
+ """
234
+
235
+ SENTIMENT_MODEL: str = "allegro/herbert-base-cased"
236
+
237
+ def __init__(self, sentiment_model: Optional[str] = None) -> None:
238
+ self._sentiment_model_id = sentiment_model or self.SENTIMENT_MODEL
239
+ self._sentiment_pipe = None
240
+
241
+ def _load(self) -> None:
242
+ if self._sentiment_pipe is None:
243
+ self._sentiment_pipe = pipeline(
244
+ "text-classification",
245
+ model=self._sentiment_model_id,
246
+ return_all_scores=False,
247
+ )
248
+
249
+ @property
250
+ def name(self) -> str:
251
+ return "herbert"
252
+
253
+ @property
254
+ def supported_languages(self) -> List[str]:
255
+ return ["pl"]
256
+
257
+ def analyze_sentiment(self, text: str) -> Dict[str, Any]:
258
+ self._load()
259
+ raw = self._sentiment_pipe(text)
260
+ result = _extract_pipeline_result(self.name, raw)
261
+ return {
262
+ "label": _normalize_sentiment_label(result.get("label", "")),
263
+ "score": float(result.get("score", 0.0)),
264
+ }
265
+
266
+
267
+ class NorBERTAdapter(ModelAdapter):
268
+ """
269
+ Adapter dla języka norweskiego.
270
+ sentyment: cardiffnlp/twitter-xlm-roberta-base-sentiment-multilingual
271
+ (wielojęzyczny XLM-RoBERTa Cardiff, obsługuje norweski)
272
+ fake news: współdzielony BART
273
+ """
274
+
275
+ # cardiffnlp/twitter-xlm-roberta-base-sentiment-multilingual —
276
+ # wielojęzyczny model Cardiff fine-tuned na sentymencie (obsługuje norweski),
277
+ # standardowa architektura XLM-RoBERTa kompatybilna z pipeline("text-classification").
278
+ SENTIMENT_MODEL: str = "cardiffnlp/twitter-xlm-roberta-base-sentiment-multilingual"
279
+
280
+ def __init__(self, sentiment_model: Optional[str] = None) -> None:
281
+ self._sentiment_model_id = sentiment_model or self.SENTIMENT_MODEL
282
+ self._sentiment_pipe = None
283
+
284
+ def _load(self) -> None:
285
+ if self._sentiment_pipe is None:
286
+ self._sentiment_pipe = pipeline(
287
+ "text-classification",
288
+ model=self._sentiment_model_id,
289
+ return_all_scores=False,
290
+ )
291
+
292
+ @property
293
+ def name(self) -> str:
294
+ return "norbert"
295
+
296
+ @property
297
+ def supported_languages(self) -> List[str]:
298
+ return ["no"]
299
+
300
+ def analyze_sentiment(self, text: str) -> Dict[str, Any]:
301
+ self._load()
302
+ raw = self._sentiment_pipe(text)
303
+ result = _extract_pipeline_result(self.name, raw)
304
+ return {
305
+ "label": _normalize_sentiment_label(result.get("label", "")),
306
+ "score": float(result.get("score", 0.0)),
307
+ }
308
+
309
+
310
+ # ---------------------------------------------------------------------------
311
+ # Rejestr adapterów — lazy initialization
312
+ # ---------------------------------------------------------------------------
313
+
314
+ _REGISTRY: Dict[str, ModelAdapter] = {}
315
+ _active_adapter: Optional[ModelAdapter] = None
316
+ executor = ThreadPoolExecutor(max_workers=3)
317
+ _BUILTIN_ADAPTERS = {"roberta", "xlm-roberta", "herbert", "norbert"}
318
+
319
+
320
+ def _init_registry() -> None:
321
+ if _REGISTRY:
322
+ return
323
+ _REGISTRY["roberta"] = RoBERTaAdapter()
324
+ _REGISTRY["xlm-roberta"] = XLMRoBERTaAdapter()
325
+ _REGISTRY["herbert"] = HerBERTAdapter()
326
+ _REGISTRY["norbert"] = NorBERTAdapter()
327
+
328
+
329
+ def get_adapter(name: str) -> ModelAdapter:
330
+ _init_registry()
331
+ if name not in _REGISTRY:
332
+ raise ValueError(
333
+ f"Nieznany adapter: '{name}'. Dostępne: {list(_REGISTRY.keys())}"
334
+ )
335
+ return _REGISTRY[name]
336
+
337
+
338
+ def set_active_adapter(name: str) -> None:
339
+ global _active_adapter
340
+ _active_adapter = get_adapter(name)
341
+
342
+
343
+ def register_adapter(adapter: ModelAdapter) -> None:
344
+ _init_registry()
345
+ _REGISTRY[adapter.name] = adapter
346
+
347
+
348
+ def _get_active_adapter() -> ModelAdapter:
349
+ global _active_adapter
350
+ if _active_adapter is None:
351
+ _active_adapter = get_adapter("roberta")
352
+ return _active_adapter
353
+
354
+
355
+ # ---------------------------------------------------------------------------
356
+ # Publiczne API
357
+ # ---------------------------------------------------------------------------
358
+
359
+ def analyze_news(
360
+ text: str,
361
+ lang: str = "pl",
362
+ adapter: Optional[ModelAdapter] = None,
363
+ ) -> dict:
364
+ """
365
+ Analizuje pojedynczy tekst pod kątem sentymentu i fake news.
366
+
367
+ Returns:
368
+ {"sentiment": str, "fake_probability": float, "sentiment_score": float}
369
+ """
370
+ _neutral = SENTIMENT_MAP["neutral"].get(lang, "Neutral")
371
+
372
+ if not text or len(text.strip()) < 10:
373
+ return {"sentiment": _neutral, "fake_probability": 0.0, "sentiment_score": 0.0}
374
+
375
+ _adapter = adapter or _get_active_adapter()
376
+
377
+ try:
378
+ sentiment_result = _adapter.analyze_sentiment(text)
379
+ fake_result = _adapter.analyze_fake_news(text)
380
+ except Exception:
381
+ return {"sentiment": _neutral, "fake_probability": 0.0, "sentiment_score": 0.0}
382
+
383
+ label = sentiment_result.get("label", "neutral")
384
+ sentiment_translated = SENTIMENT_MAP.get(label, {}).get(lang, _neutral)
385
+
386
+ fake_score = 0.0
387
+ for lbl, score in zip(fake_result["labels"], fake_result["scores"]):
388
+ if lbl == "fake":
389
+ fake_score = score
390
+ break
391
+
392
+ return {
393
+ "sentiment": sentiment_translated,
394
+ "fake_probability": round(fake_score * 100, 2),
395
+ "sentiment_score": round(float(sentiment_result.get("score", 0.0)), 2),
396
+ }
397
+
398
+
399
+ def analyze_news_batch(
400
+ texts: List[str],
401
+ lang: str = "pl",
402
+ adapter: Optional[ModelAdapter] = None,
403
+ ) -> List[Dict[str, Any]]:
404
+ """Analiza wielu tekstów w trybie batch z użyciem ThreadPoolExecutor."""
405
+ if not texts:
406
+ return []
407
+
408
+ _adapter = adapter or _get_active_adapter()
409
+ results: List[Dict[str, Any]] = []
410
+ batch_size = 3
411
+
412
+ batches = [texts[i:i + batch_size] for i in range(0, len(texts), batch_size)]
413
+
414
+ for batch in batches:
415
+ futures = [
416
+ executor.submit(analyze_news, text, lang, _adapter)
417
+ for text in batch
418
+ ]
419
+ for future in futures:
420
+ try:
421
+ results.append(future.result())
422
+ except Exception:
423
+ _neutral = SENTIMENT_MAP["neutral"].get(lang, "Neutral")
424
+ results.append(
425
+ {"sentiment": _neutral, "fake_probability": 0.0, "sentiment_score": 0.0}
426
+ )
427
+
428
+ return results
429
+
430
+
431
+ def analyze_news_single(text: str, lang: str) -> Dict[str, Any]:
432
+ """Wrapper dla analizy pojedynczego tekstu (używany w batch processing)."""
433
+ return analyze_news(text, lang)
app/routes/__init__.py ADDED
File without changes
app/routes/compare.py ADDED
@@ -0,0 +1,372 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Endpoint porównujący wyniki wszystkich adapterów NLP dla jednego tekstu.
3
+
4
+ Architektura:
5
+ - fake news: jeden wspólny wynik z _shared_bart (BART large MNLI)
6
+ - sentyment: każdy z 4 adapterów używa własnego modelu
7
+
8
+ Zwraca:
9
+ - shared_fake_probability: float — wynik BART (jeden dla wszystkich)
10
+ - results: lista wyników sentymentu per adapter z czasem inferencji
11
+ - explanation: wyjaśnienie różnic wygenerowane przez Claude API
12
+ """
13
+
14
+ import time
15
+ from typing import List
16
+
17
+ import anthropic
18
+ from fastapi import APIRouter, HTTPException
19
+ from pydantic import BaseModel
20
+
21
+ from ..config import ANTHROPIC_API_KEY, SENTIMENT_MAP
22
+ from ..nlp_service import get_adapter, analyze_fake_news_shared
23
+
24
+ router = APIRouter()
25
+
26
+ ADAPTER_NAMES = ["roberta", "xlm-roberta", "norbert", "herbert"]
27
+
28
+ LANG_NAMES = {
29
+ "pl": "polski",
30
+ "en": "angielski",
31
+ "no": "norweski",
32
+ }
33
+
34
+ # Język odpowiedzi Claude zależny od lang w żądaniu
35
+ RESPONSE_LANG = {
36
+ "pl": "polskim",
37
+ "en": "English",
38
+ "no": "norsk",
39
+ }
40
+
41
+ # Instrukcja do Claude w danym języku
42
+ EXPLAIN_INSTRUCTION = {
43
+ "pl": (
44
+ "Wyjaśnij w 3–4 zdaniach po polsku dlaczego modele oceniają sentyment tego tekstu tak jak oceniają. "
45
+ "Uwzględnij: (1) czy tekst jest w języku {lang_name} i jak to wpływa na modele trenowane na innych językach, "
46
+ "(2) co powoduje ewentualne rozbieżności między modelami, "
47
+ "(3) który model jest prawdopodobnie najbardziej trafny dla tego języka i dlaczego. "
48
+ "Pisz czystym tekstem — bez formatowania markdown, bez gwiazdek, bez nagłówków, bez list numerowanych."
49
+ ),
50
+ "en": (
51
+ "Explain in 3–4 sentences in English why the models assess the sentiment of this text the way they do. "
52
+ "Cover: (1) whether the text is in {lang_name} and how that affects models trained on other languages, "
53
+ "(2) what causes any disagreements between the models, "
54
+ "(3) which model is likely most accurate for this language and why. "
55
+ "Write in plain text — no markdown, no asterisks, no headings, no numbered lists."
56
+ ),
57
+ "no": (
58
+ "Forklar på norsk i 3–4 setninger hvorfor modellene vurderer sentimentet til denne teksten slik de gjør. "
59
+ "Ta med: (1) om teksten er på {lang_name} og hvordan det påvirker modeller trent på andre språk, "
60
+ "(2) hva som forårsaker eventuelle uenigheter mellom modellene, "
61
+ "(3) hvilken modell som sannsynligvis er mest nøyaktig for dette språket og hvorfor. "
62
+ "Skriv i ren tekst — ingen markdown, ingen stjerner, ingen overskrifter, ingen nummererte lister."
63
+ ),
64
+ }
65
+
66
+ ADAPTER_DESCRIPTIONS = {
67
+ "roberta": "RoBERTa (trenowany na angielskich tweetach)",
68
+ "xlm-roberta": "XLM-RoBERTa (wielojęzyczny, pl/en/no)",
69
+ "norbert": "NorBERT 3 (norweski model bazowy)",
70
+ "herbert": "HerBERT (polski model bazowy)",
71
+ }
72
+
73
+
74
+ CHART_COMMENT_PROMPT = {
75
+ "pl": (
76
+ "Na podstawie poniższych danych fake-news probability (0–100%) dla {n} źródeł mediów "
77
+ "napisz dokładnie 2–3 zdania komentarza analitycznego po polsku. "
78
+ "Wskaż źródło z najwyższym i najniższym wynikiem oraz co to może oznaczać dla czytelnika. "
79
+ "Bądź zwięzły i obiektywny. Pisz czystym tekstem — bez markdown, bez gwiazdek, bez list.\n\n"
80
+ "Dane:\n{rows}"
81
+ ),
82
+ "en": (
83
+ "Based on the following fake-news probability data (0–100%) for {n} media sources "
84
+ "write exactly 2–3 sentences of analytical commentary in English. "
85
+ "Identify the source with the highest and lowest score and what this might mean for readers. "
86
+ "Be concise and objective. Write in plain text — no markdown, no asterisks, no lists.\n\n"
87
+ "Data:\n{rows}"
88
+ ),
89
+ "no": (
90
+ "Basert på følgende fake-news sannsynlighetsdata (0–100 %) for {n} mediekilder "
91
+ "skriv nøyaktig 2–3 setninger med analytisk kommentar på norsk. "
92
+ "Pek på kilden med høyest og lavest score og hva dette kan bety for leseren. "
93
+ "Vær kortfattet og objektiv. Skriv i ren tekst — ingen markdown, ingen stjerner, ingen lister.\n\n"
94
+ "Data:\n{rows}"
95
+ ),
96
+ }
97
+
98
+
99
+ class ChartCommentRequest(BaseModel):
100
+ bars: list # [{"label": "BBC", "value": 23.45}, ...]
101
+ lang: str = "pl"
102
+
103
+
104
+ class ChartCommentResponse(BaseModel):
105
+ comment: str
106
+
107
+
108
+ @router.post("/ai-chart-comment", response_model=ChartCommentResponse)
109
+ def ai_chart_comment(request: ChartCommentRequest):
110
+ """Generuje krótki komentarz analityczny Claude do wykresu fake-news probability."""
111
+ if not request.bars:
112
+ raise HTTPException(status_code=422, detail="Brak danych wykresu.")
113
+
114
+ lang = request.lang if request.lang in CHART_COMMENT_PROMPT else "pl"
115
+
116
+ # Buduj czytelny opis danych (posortowany malejąco po value)
117
+ sorted_bars = sorted(request.bars, key=lambda b: b.get("value", 0), reverse=True)
118
+ rows = "\n".join(
119
+ f" {b['label']}: {b.get('value', 0):.2f}%"
120
+ for b in sorted_bars
121
+ if b.get("label")
122
+ )
123
+ prompt = CHART_COMMENT_PROMPT[lang].format(n=len(sorted_bars), rows=rows)
124
+
125
+ if not ANTHROPIC_API_KEY:
126
+ return ChartCommentResponse(comment="[Brak klucza ANTHROPIC_API_KEY]")
127
+
128
+ try:
129
+ client = anthropic.Anthropic(api_key=ANTHROPIC_API_KEY)
130
+ message = client.messages.create(
131
+ model="claude-sonnet-4-20250514",
132
+ max_tokens=300,
133
+ messages=[{"role": "user", "content": prompt}],
134
+ )
135
+ comment = message.content[0].text.strip() if message.content else ""
136
+ except Exception as e:
137
+ comment = f"[Błąd Claude API: {e}]"
138
+
139
+ return ChartCommentResponse(comment=comment)
140
+
141
+
142
+ EMOTION_COMMENT_PROMPT = {
143
+ "pl": (
144
+ "Na podstawie danych o emocjach w artykułach: "
145
+ "Pozytywne: {positive}, Neutralne: {neutral}, Negatywne: {negative} "
146
+ "(łącznie {total} artykułów) — napisz 2–3 zdania komentarza analitycznego po polsku. "
147
+ "Co dominujący sentyment mówi o obecnym klimacie medialnym? "
148
+ "Bądź zwięzły i obiektywny. Pisz czystym tekstem — bez markdown, bez gwiazdek, bez list."
149
+ ),
150
+ "en": (
151
+ "Based on emotion data from articles: "
152
+ "Positive: {positive}, Neutral: {neutral}, Negative: {negative} "
153
+ "(total {total} articles) — write 2–3 sentences of analytical commentary in English. "
154
+ "What does the dominant sentiment say about the current media climate? "
155
+ "Be concise and objective. Write in plain text — no markdown, no asterisks, no lists."
156
+ ),
157
+ "no": (
158
+ "Basert på emosjonelle data fra artikler: "
159
+ "Positive: {positive}, Nøytrale: {neutral}, Negative: {negative} "
160
+ "(totalt {total} artikler) — skriv 2–3 setninger med analytisk kommentar på norsk. "
161
+ "Hva sier det dominerende sentimentet om det nåværende mediaklimaet? "
162
+ "Vær kortfattet og objektiv. Skriv i ren tekst — ingen markdown, ingen stjerner, ingen lister."
163
+ ),
164
+ }
165
+
166
+
167
+ class EmotionCommentRequest(BaseModel):
168
+ positive: int = 0
169
+ neutral: int = 0
170
+ negative: int = 0
171
+ lang: str = "pl"
172
+
173
+
174
+ class EmotionCommentResponse(BaseModel):
175
+ comment: str
176
+
177
+
178
+ @router.post("/ai-emotion-comment", response_model=EmotionCommentResponse)
179
+ def ai_emotion_comment(request: EmotionCommentRequest):
180
+ """Generuje krótki komentarz analityczny Claude do wykresu emocji."""
181
+ total = request.positive + request.neutral + request.negative
182
+ if total == 0:
183
+ raise HTTPException(status_code=422, detail="Brak danych emocji.")
184
+
185
+ lang = request.lang if request.lang in EMOTION_COMMENT_PROMPT else "pl"
186
+ prompt = EMOTION_COMMENT_PROMPT[lang].format(
187
+ positive=request.positive,
188
+ neutral=request.neutral,
189
+ negative=request.negative,
190
+ total=total,
191
+ )
192
+
193
+ if not ANTHROPIC_API_KEY:
194
+ return EmotionCommentResponse(comment="[Brak klucza ANTHROPIC_API_KEY]")
195
+
196
+ try:
197
+ client = anthropic.Anthropic(api_key=ANTHROPIC_API_KEY)
198
+ message = client.messages.create(
199
+ model="claude-sonnet-4-20250514",
200
+ max_tokens=300,
201
+ messages=[{"role": "user", "content": prompt}],
202
+ )
203
+ comment = message.content[0].text.strip() if message.content else ""
204
+ except Exception as e:
205
+ comment = f"[Błąd Claude API: {e}]"
206
+
207
+ return EmotionCommentResponse(comment=comment)
208
+
209
+
210
+ class CompareRequest(BaseModel):
211
+ text: str
212
+ title: str = ""
213
+ source: str = ""
214
+ lang: str = "pl"
215
+
216
+
217
+ class AdapterResult(BaseModel):
218
+ adapter: str
219
+ sentiment: str
220
+ sentiment_score: float
221
+ inference_time_ms: float
222
+
223
+
224
+ class CompareResponse(BaseModel):
225
+ shared_fake_probability: float
226
+ results: List[AdapterResult]
227
+ explanation: str
228
+
229
+
230
+ def _sentiment_only_with_timing(adapter_name: str, text: str, lang: str) -> AdapterResult:
231
+ """Uruchamia tylko analizę sentymentu i mierzy jej czas."""
232
+ adapter = get_adapter(adapter_name)
233
+ neutral = SENTIMENT_MAP["neutral"].get(lang, "Neutral")
234
+
235
+ if not text or len(text.strip()) < 10:
236
+ print(f"[DEBUG] {adapter_name}: tekst za krótki, zwracam neutral/0.0")
237
+ return AdapterResult(
238
+ adapter=adapter_name,
239
+ sentiment=neutral,
240
+ sentiment_score=0.0,
241
+ inference_time_ms=0.0,
242
+ )
243
+
244
+ start = time.perf_counter()
245
+ try:
246
+ sentiment_result = adapter.analyze_sentiment(text)
247
+ except Exception as exc:
248
+ elapsed = (time.perf_counter() - start) * 1000
249
+ print(f"[DEBUG] {adapter_name}: WYJĄTEK w analyze_sentiment: {type(exc).__name__}: {exc}")
250
+ return AdapterResult(
251
+ adapter=adapter_name,
252
+ sentiment=neutral,
253
+ sentiment_score=0.0,
254
+ inference_time_ms=round(elapsed, 1),
255
+ )
256
+ elapsed_ms = (time.perf_counter() - start) * 1000
257
+
258
+ print(f"[DEBUG] {adapter_name}: sentiment_result = {sentiment_result}")
259
+
260
+ raw_label = sentiment_result.get("label", "neutral")
261
+ raw_score = sentiment_result.get("score", None)
262
+
263
+ print(f"[DEBUG] {adapter_name}: raw_label={raw_label!r} raw_score={raw_score!r} type(score)={type(raw_score)}")
264
+
265
+ sentiment_translated = SENTIMENT_MAP.get(raw_label, {}).get(lang, neutral)
266
+ score_float = round(float(raw_score) if raw_score is not None else 0.0, 4)
267
+
268
+ print(f"[DEBUG] {adapter_name}: → sentiment={sentiment_translated!r} score_float={score_float} time={elapsed_ms:.1f}ms")
269
+
270
+ return AdapterResult(
271
+ adapter=adapter_name,
272
+ sentiment=sentiment_translated,
273
+ sentiment_score=score_float,
274
+ inference_time_ms=round(elapsed_ms, 1),
275
+ )
276
+
277
+
278
+ def _build_prompt(
279
+ request: CompareRequest,
280
+ results: List[AdapterResult],
281
+ shared_fake_probability: float,
282
+ ) -> str:
283
+ lang_name = LANG_NAMES.get(request.lang, request.lang)
284
+ text_snippet = request.text[:600] + ("..." if len(request.text) > 600 else "")
285
+
286
+ header_parts = []
287
+ if request.title:
288
+ header_parts.append(f"Tytuł artykułu: {request.title}")
289
+ if request.source:
290
+ header_parts.append(f"Źródło: {request.source}")
291
+ header_parts.append(f"Język tekstu: {lang_name}")
292
+ header_parts.append(f"Prawdopodobieństwo fake news (BART, wspólne dla wszystkich modeli): {shared_fake_probability:.1f}%")
293
+ header_str = "\n".join(header_parts)
294
+
295
+ rows = "\n".join(
296
+ f" {ADAPTER_DESCRIPTIONS[r.adapter]}: {r.sentiment} (pewność: {r.sentiment_score * 100:.0f}%, czas: {r.inference_time_ms:.0f} ms)"
297
+ for r in results
298
+ )
299
+
300
+ sentiments = [r.sentiment for r in results]
301
+ unique_sentiments = set(sentiments)
302
+ agreement = (
303
+ "zgodne" if len(unique_sentiments) == 1
304
+ else f"różne ({', '.join(unique_sentiments)})"
305
+ )
306
+
307
+ lang = request.lang
308
+ response_lang = RESPONSE_LANG.get(lang, "polskim")
309
+ instruction = EXPLAIN_INSTRUCTION.get(lang, EXPLAIN_INSTRUCTION["pl"]).format(
310
+ lang_name=lang_name
311
+ )
312
+
313
+ return f"""Odpowiedz w języku: {response_lang}.
314
+
315
+ {header_str}
316
+
317
+ Fragment tekstu:
318
+ {text_snippet}
319
+
320
+ Wyniki analizy sentymentu:
321
+ {rows}
322
+
323
+ Zgodność modeli: {agreement}
324
+
325
+ {instruction}"""
326
+
327
+
328
+ @router.post("/compare", response_model=CompareResponse)
329
+ def compare_models(request: CompareRequest):
330
+ """
331
+ Analizuje tekst przez wszystkie 4 adaptery NLP:
332
+ - fake news: jeden wynik ze współdzielonego BART
333
+ - sentyment: każdy adapter osobno z pomiarem czasu
334
+ Zwraca wyniki + wyjaśnienie Claude API.
335
+ """
336
+ if not request.text or len(request.text.strip()) < 10:
337
+ raise HTTPException(status_code=422, detail="Tekst jest za krótki (min. 10 znaków).")
338
+
339
+ # Fake news — jeden wspólny wynik z BART (uruchamiamy raz)
340
+ try:
341
+ shared_fake_probability = analyze_fake_news_shared(request.text)
342
+ except Exception as e:
343
+ shared_fake_probability = 0.0
344
+
345
+ # Sentyment — każdy adapter osobno z pomiarem czasu
346
+ results: List[AdapterResult] = [
347
+ _sentiment_only_with_timing(name, request.text, request.lang)
348
+ for name in ADAPTER_NAMES
349
+ ]
350
+
351
+ # Wyjaśnienie Claude
352
+ explanation = ""
353
+ if ANTHROPIC_API_KEY:
354
+ try:
355
+ client = anthropic.Anthropic(api_key=ANTHROPIC_API_KEY)
356
+ prompt = _build_prompt(request, results, shared_fake_probability)
357
+ message = client.messages.create(
358
+ model="claude-sonnet-4-20250514",
359
+ max_tokens=800,
360
+ messages=[{"role": "user", "content": prompt}],
361
+ )
362
+ explanation = message.content[0].text if message.content else ""
363
+ except Exception as e:
364
+ explanation = f"[Błąd Claude API: {str(e)}]"
365
+ else:
366
+ explanation = "[Brak klucza ANTHROPIC_API_KEY — wyjaśnienie niedostępne]"
367
+
368
+ return CompareResponse(
369
+ shared_fake_probability=shared_fake_probability,
370
+ results=results,
371
+ explanation=explanation,
372
+ )
app/routes/misc.py ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Endpointy związane z obsługą dostępnych źródeł wiadomości oraz narzędziami deweloperskimi.
3
+ """
4
+
5
+ from pathlib import Path
6
+ from typing import Optional
7
+
8
+ from fastapi import APIRouter, Query
9
+
10
+ from ..config import NEWS_FEEDS
11
+ from ..benchmark import run_benchmark, export_json, export_csv, _summary_only
12
+
13
+ router = APIRouter()
14
+
15
+ # Katalog, do którego endpoint zapisuje artefakty benchmarku
16
+ _BENCHMARK_OUT = Path("benchmark_results")
17
+
18
+
19
+ @router.get("/sources")
20
+ def get_sources():
21
+ return list(NEWS_FEEDS.keys())
22
+
23
+
24
+ @router.get("/benchmark")
25
+ def benchmark(
26
+ adapters: Optional[str] = Query(
27
+ default=None,
28
+ description="Przecinkowa lista adapterów: roberta,xlm-roberta,norbert",
29
+ ),
30
+ langs: Optional[str] = Query(
31
+ default=None,
32
+ description="Przecinkowa lista języków: en,pl,no",
33
+ ),
34
+ save: bool = Query(
35
+ default=True,
36
+ description="Czy zapisać wyniki do JSON i CSV w katalogu benchmark_results/",
37
+ ),
38
+ full: bool = Query(
39
+ default=False,
40
+ description="Czy zwrócić szczegółowe wyniki per_text (domyślnie tylko podsumowanie)",
41
+ ),
42
+ ):
43
+ """
44
+ Uruchamia benchmark NLP i zwraca wyniki.
45
+
46
+ Czas odpowiedzi zależy od liczby adapterów i języków — może wynosić kilkadziesiąt sekund
47
+ przy pierwszym uruchomieniu (lazy-loading modeli HuggingFace).
48
+
49
+ Przykłady:
50
+ - GET /benchmark
51
+ - GET /benchmark?adapters=roberta,xlm-roberta&langs=en,pl
52
+ - GET /benchmark?full=true&save=false
53
+ """
54
+ adapter_names = [a.strip() for a in adapters.split(",")] if adapters else None
55
+ lang_list = [l.strip() for l in langs.split(",")] if langs else None
56
+
57
+ results = run_benchmark(adapter_names=adapter_names, langs=lang_list)
58
+
59
+ if save:
60
+ export_json(results, _BENCHMARK_OUT / "benchmark.json")
61
+ export_csv(results, _BENCHMARK_OUT / "benchmark.csv")
62
+
63
+ return {
64
+ "results": results if full else _summary_only(results),
65
+ "saved": save,
66
+ "output_dir": str(_BENCHMARK_OUT) if save else None,
67
+ }
68
+
app/routes/news.py ADDED
@@ -0,0 +1,266 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Endpointy związane z pobieraniem, analizą i strumieniowaniem wiadomości.
3
+ """
4
+
5
+ import asyncio
6
+ import json
7
+ from typing import List
8
+
9
+ from fastapi import APIRouter, HTTPException
10
+ from fastapi.responses import StreamingResponse
11
+ from fastapi_cache.decorator import cache
12
+
13
+ from ..config import CACHE_TTL, NEWS_FEEDS
14
+ from ..rss_utils import fetch_feed, clean_html, get_from_cache, set_to_cache, FeedFetchError
15
+ from ..nlp_service import analyze_news, set_active_adapter
16
+
17
+ router = APIRouter()
18
+
19
+
20
+ @router.get("/news/{source}")
21
+ def get_news(source: str, lang: str = "pl", model: str = "roberta"):
22
+ # Walidacja źródła
23
+ if source not in NEWS_FEEDS:
24
+ raise HTTPException(status_code=404, detail="Źródło nieobsługiwane")
25
+
26
+ # Ustawienie aktywnego adaptera NLP
27
+ try:
28
+ set_active_adapter(model)
29
+ except ValueError:
30
+ raise HTTPException(status_code=400, detail=f"Nieznany model: {model}")
31
+
32
+ url = NEWS_FEEDS[source]
33
+
34
+ # Pobranie RSS — FeedFetchError (404/timeout) zwraca pustą listę zamiast HTTP 500
35
+ try:
36
+ feed = fetch_feed(url)
37
+ except FeedFetchError as e:
38
+ print(f"⚠️ Feed niedostępny [{source}]: {e}")
39
+ return {"source": source, "articles": [], "error": str(e)}
40
+ except Exception as e:
41
+ raise HTTPException(status_code=500, detail=f"Błąd pobierania newsów: {str(e)}")
42
+
43
+ if not feed.entries:
44
+ return {"source": source, "articles": []}
45
+
46
+ MAX_ARTICLES = 5
47
+ articles: List[dict] = []
48
+
49
+ # Przetwarzanie i analiza artykułów
50
+ for entry in feed.entries[:MAX_ARTICLES]:
51
+ summary = clean_html(entry.get("summary", "Brak opisu"))
52
+ text_to_analyze = (summary or "").strip() or entry.get("title", "")
53
+ analysis = analyze_news(text_to_analyze, lang)
54
+
55
+ articles.append({
56
+ "title": entry.get("title", "Bez tytułu"),
57
+ "link": entry.get("link", ""),
58
+ "summary": summary,
59
+ "published": entry.get("published", "Brak daty"),
60
+ "source": source,
61
+ "model": model,
62
+ **analysis
63
+ })
64
+
65
+ return {"source": source, "articles": articles}
66
+
67
+
68
+ # Strumieniowanie newsów przez SSE
69
+ @router.get("/stream-news/{source}")
70
+ async def stream_news(source: str, lang: str = "pl", model: str = "roberta"):
71
+ if source not in NEWS_FEEDS:
72
+ raise HTTPException(status_code=404, detail="Źródło nieobsługiwane")
73
+
74
+ # Walidacja i ustawienie adaptera przed uruchomieniem generatora
75
+ try:
76
+ set_active_adapter(model)
77
+ except ValueError:
78
+ raise HTTPException(status_code=400, detail=f"Nieznany model: {model}")
79
+
80
+ async def event_generator():
81
+ # Próba pobrania RSS z cache
82
+ try:
83
+ cached = get_from_cache(source)
84
+ if cached:
85
+ feed = cached
86
+ else:
87
+ feed = fetch_feed(NEWS_FEEDS[source])
88
+ set_to_cache(source, feed)
89
+ except FeedFetchError as e:
90
+ print(f"⚠️ Feed niedostępny [{source}]: {e}")
91
+ yield f"event: backend_error\ndata: {json.dumps({'message': str(e), 'status': e.status_code})}\n\n"
92
+ yield f"event: done\ndata: {json.dumps({'count': 0})}\n\n"
93
+ return
94
+ except Exception as e:
95
+ yield f"event: backend_error\ndata: {json.dumps({'message': str(e)})}\n\n"
96
+ yield f"event: done\ndata: {json.dumps({'count': 0})}\n\n"
97
+ return
98
+
99
+ entries = feed.entries or []
100
+ MAX_ARTICLES = 5
101
+ to_send = entries[:MAX_ARTICLES]
102
+
103
+ # Metadane dla klienta
104
+ yield f"event: meta\ndata: {json.dumps({'total': len(to_send)})}\n\n"
105
+
106
+ sent = 0
107
+ for entry in to_send:
108
+ summary = clean_html(entry.get("summary", "Brak opisu"))
109
+ text_to_analyze = (summary or "").strip() or entry.get("title", "")
110
+
111
+ # Analiza NLP uruchamiana w executorze (CPU-bound)
112
+ loop = asyncio.get_event_loop()
113
+ analysis = await loop.run_in_executor(
114
+ None, lambda: analyze_news(text_to_analyze, lang)
115
+ )
116
+
117
+ article = {
118
+ "title": entry.get("title", "Bez tytułu"),
119
+ "link": entry.get("link", ""),
120
+ "summary": summary,
121
+ "published": entry.get("published", "Brak daty"),
122
+ "source": source,
123
+ "model": model,
124
+ **analysis,
125
+ }
126
+
127
+ yield f"data: {json.dumps(article, ensure_ascii=False)}\n\n"
128
+ sent += 1
129
+ await asyncio.sleep(0.3)
130
+
131
+ yield f"event: done\ndata: {json.dumps({'count': sent})}\n\n"
132
+
133
+ return StreamingResponse(
134
+ event_generator(),
135
+ media_type="text/event-stream",
136
+ headers={
137
+ "Content-Type": "text/event-stream",
138
+ "Access-Control-Allow-Origin": "*",
139
+ "Access-Control-Allow-Headers": "*",
140
+ "Cache-Control": "no-cache",
141
+ "X-Accel-Buffering": "no",
142
+ },
143
+ )
144
+
145
+
146
+ @router.get("/api/charts/summary")
147
+ @cache(expire=CACHE_TTL)
148
+ async def get_charts_summary(lang: str = "pl"):
149
+ """
150
+ Szybkie statystyki zbiorcze dla wszystkich źródeł
151
+ (uproszczona analiza oparta na tytułach).
152
+ """
153
+ sources = [
154
+ "BBC", "CNN", "NYTimes", "Guardian", "AlJazeera",
155
+ "PolsatNews", "Money", "Bankier", "SpidersWeb", "GazetaPrawna"
156
+ ]
157
+
158
+ summary_data = {}
159
+
160
+ for source in sources:
161
+ if source not in NEWS_FEEDS:
162
+ continue
163
+
164
+ try:
165
+ feed = fetch_feed(NEWS_FEEDS[source])
166
+
167
+ if not feed.entries:
168
+ summary_data[source] = {
169
+ "count": 0,
170
+ "emotions": {"Pozytywne": 0, "Negatywne": 0, "Neutralne": 0}
171
+ }
172
+ continue
173
+
174
+ # Analiza tylko kilku tytułów (szybko)
175
+ articles = feed.entries[:3]
176
+ emotion_counts = {"Pozytywne": 0, "Negatywne": 0, "Neutralne": 0}
177
+
178
+ for entry in articles:
179
+ title = entry.get("title", "").lower()
180
+ if any(word in title for word in ["good", "positive", "gain", "up", "success", "dobry", "wzrost", "zysk"]):
181
+ emotion_counts["Pozytywne"] += 1
182
+ elif any(word in title for word in ["bad", "negative", "fall", "down", "loss", "crisis", "zły", "spadek", "kryzys"]):
183
+ emotion_counts["Negatywne"] += 1
184
+ else:
185
+ emotion_counts["Neutralne"] += 1
186
+
187
+ summary_data[source] = {
188
+ "count": len(feed.entries),
189
+ "analyzed": len(articles),
190
+ "emotions": emotion_counts,
191
+ "latest_title": articles[0].get("title", "")[:50] if articles else ""
192
+ }
193
+
194
+ except Exception:
195
+ summary_data[source] = {
196
+ "count": 0,
197
+ "emotions": {"Pozytywne": 0, "Negatywne": 0, "Neutralne": 0}
198
+ }
199
+
200
+ return {
201
+ "summary": summary_data,
202
+ "total_sources": len(summary_data),
203
+ "cache_ttl": CACHE_TTL
204
+ }
205
+
206
+
207
+ @router.get("/emotion-stats/{source}")
208
+ @cache(expire=CACHE_TTL)
209
+ def get_emotion_stats(source: str, lang: str = "pl", model: str = "roberta"):
210
+ # Statystyki emocji dla jednego źródła
211
+ if source not in NEWS_FEEDS:
212
+ raise HTTPException(status_code=404, detail="Źródło nieobsługiwane")
213
+
214
+ try:
215
+ news_data = get_news(source, lang, model)
216
+ articles = news_data.get("articles", [])
217
+ except Exception as e:
218
+ raise HTTPException(status_code=500, detail=f"Błąd generowania statystyk: {str(e)}")
219
+
220
+ emotion_counts = {"Pozytywne": 0, "Negatywne": 0, "Neutralne": 0}
221
+
222
+ for article in articles:
223
+ sentiment = article.get("sentiment", "Neutralne")
224
+ if sentiment in emotion_counts:
225
+ emotion_counts[sentiment] += 1
226
+
227
+ total = len(articles)
228
+ emotion_percentages = {
229
+ emotion: (round((count / total) * 100, 2) if total > 0 else 0)
230
+ for emotion, count in emotion_counts.items()
231
+ }
232
+
233
+ return {
234
+ "source": source,
235
+ "total_articles": total,
236
+ "emotion_counts": emotion_counts,
237
+ "emotion_percentages": emotion_percentages
238
+ }
239
+
240
+
241
+ @router.get("/charts-data")
242
+ @cache(expire=CACHE_TTL)
243
+ async def get_all_charts_data(lang: str = "pl"):
244
+ """
245
+ Kompatybilność ze starszym frontendem – agreguje dane ze wszystkich źródeł.
246
+ """
247
+ sources = [
248
+ "BBC", "CNN", "NYTimes", "Guardian", "AlJazeera",
249
+ "PolsatNews", "Money", "Bankier", "SpidersWeb", "GazetaPrawna"
250
+ ]
251
+
252
+ async def fetch_stats(source):
253
+ try:
254
+ return {source: await get_emotion_stats(source, lang)}
255
+ except Exception:
256
+ return {source: None}
257
+
258
+ tasks = [fetch_stats(source) for source in sources]
259
+ results = await asyncio.gather(*tasks, return_exceptions=True)
260
+
261
+ all_data = {}
262
+ for result in results:
263
+ if isinstance(result, dict):
264
+ all_data.update(result)
265
+
266
+ return {"charts": all_data}
app/routes/saved.py ADDED
@@ -0,0 +1,34 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Endpointy związane z zapisywaniem i zarządzaniem zapisanymi artykułami.
3
+ """
4
+
5
+ from fastapi import APIRouter, HTTPException, Request
6
+
7
+ from ..storage import read_all, append_article, delete_by_title
8
+ from ..models import Article
9
+
10
+ router = APIRouter()
11
+
12
+
13
+ @router.get("/saved-articles")
14
+ def get_saved_articles():
15
+ # Zwraca listę wszystkich zapisanych artykułów
16
+ return read_all()
17
+
18
+
19
+ @router.post("/save-article")
20
+ def save_article(article: Article):
21
+ # Zapisuje nowy artykuł do magazynu danych
22
+ return append_article(article.dict())
23
+
24
+
25
+ @router.delete("/delete-article")
26
+ async def delete_article(request: Request):
27
+ # Usuwa artykuł na podstawie tytułu przekazanego w body requestu
28
+ data = await request.json()
29
+ title = data.get("title")
30
+
31
+ if not title:
32
+ raise HTTPException(status_code=400, detail="Brak pola 'title'")
33
+
34
+ return delete_by_title(title)
app/rss_utils.py ADDED
@@ -0,0 +1,70 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Narzędzia pomocnicze do pobierania i przetwarzania kanałów RSS oraz cache w pamięci.
3
+ """
4
+
5
+ import time
6
+ import requests
7
+ import feedparser
8
+ from bs4 import BeautifulSoup
9
+ from typing import Dict, Any
10
+
11
+ from .config import CACHE_TTL_SECONDS
12
+
13
+ # Prosty cache w pamięci (key -> (value, timestamp))
14
+ _cache_data: Dict[str, tuple[Any, float]] = {}
15
+
16
+
17
+ def clean_html(text: str) -> str:
18
+ # Usuwa znaczniki HTML z treści RSS
19
+ return BeautifulSoup(text or "", "html.parser").get_text()
20
+
21
+
22
+ def get_from_cache(key: str):
23
+ # Pobiera dane z cache, jeśli nie przekroczyły TTL
24
+ entry = _cache_data.get(key)
25
+ if not entry:
26
+ return None
27
+
28
+ value, ts = entry
29
+ if time.time() - ts > CACHE_TTL_SECONDS:
30
+ del _cache_data[key]
31
+ return None
32
+
33
+ return value
34
+
35
+
36
+ def set_to_cache(key: str, value):
37
+ # Zapisuje dane do cache wraz z timestampem
38
+ _cache_data[key] = (value, time.time())
39
+
40
+
41
+ class FeedFetchError(Exception):
42
+ """Wyjątek rzucany gdy feed RSS jest niedostępny lub zwraca błąd HTTP."""
43
+ def __init__(self, url: str, status_code: int, message: str = ""):
44
+ self.url = url
45
+ self.status_code = status_code
46
+ super().__init__(message or f"Feed unavailable: HTTP {status_code} for {url}")
47
+
48
+
49
+ def fetch_feed(url: str):
50
+ """
51
+ Pobiera i parsuje kanał RSS z ustawionym User-Agent.
52
+ Rzuca FeedFetchError jeśli serwer zwróci błąd HTTP (4xx/5xx),
53
+ co pozwala wywołującemu zwrócić pustą listę zamiast HTTP 500.
54
+ """
55
+ headers = {
56
+ "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
57
+ "Accept": "application/rss+xml, application/xml, text/xml, */*",
58
+ }
59
+ try:
60
+ resp = requests.get(url, timeout=7, headers=headers)
61
+ except requests.exceptions.Timeout:
62
+ raise FeedFetchError(url, 408, f"Timeout fetching feed: {url}")
63
+ except requests.exceptions.ConnectionError as e:
64
+ raise FeedFetchError(url, 503, f"Connection error for {url}: {e}")
65
+
66
+ if resp.status_code >= 400:
67
+ raise FeedFetchError(url, resp.status_code,
68
+ f"HTTP {resp.status_code} for feed: {url}")
69
+
70
+ return feedparser.parse(resp.content)
app/storage.py ADDED
@@ -0,0 +1,60 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Warstwa dostępu do danych dla zapisanych artykułów (plik JSON).
3
+ """
4
+
5
+ import json
6
+ import threading
7
+ from fastapi import HTTPException
8
+
9
+ from .config import SAVED_FILE
10
+
11
+ # Blokada wątków dla operacji zapisu/odczytu
12
+ _lock = threading.Lock()
13
+
14
+
15
+ def ensure_file():
16
+ # Tworzy plik danych, jeśli nie istnieje
17
+ if not SAVED_FILE.exists():
18
+ SAVED_FILE.write_text("[]", encoding="utf-8")
19
+
20
+
21
+ def read_all():
22
+ # Zwraca wszystkie zapisane artykuły
23
+ ensure_file()
24
+ try:
25
+ return json.loads(SAVED_FILE.read_text(encoding="utf-8"))
26
+ except json.JSONDecodeError:
27
+ return []
28
+
29
+
30
+ def append_article(article: dict):
31
+ # Dodaje nowy artykuł do pliku JSON
32
+ ensure_file()
33
+ try:
34
+ with _lock, open(SAVED_FILE, "r+", encoding="utf-8") as f:
35
+ data = json.load(f)
36
+ data.append(article)
37
+ f.seek(0)
38
+ json.dump(data, f, ensure_ascii=False, indent=4)
39
+
40
+ return {"message": "Artykuł zapisany pomyślnie."}
41
+ except Exception as e:
42
+ raise HTTPException(status_code=500, detail=str(e))
43
+
44
+
45
+ def delete_by_title(title: str):
46
+ # Usuwa artykuł na podstawie tytułu
47
+ ensure_file()
48
+ try:
49
+ with _lock, open(SAVED_FILE, "r+", encoding="utf-8") as f:
50
+ saved = json.load(f)
51
+ new_saved = [
52
+ a for a in saved if a.get("title") != title
53
+ ]
54
+ f.seek(0)
55
+ f.truncate()
56
+ json.dump(new_saved, f, ensure_ascii=False, indent=4)
57
+
58
+ return {"message": "Artykuł usunięty."}
59
+ except Exception as e:
60
+ raise HTTPException(status_code=500, detail=str(e))
requirements.txt ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Zależności backendu aplikacji TruthScan
2
+
3
+ # Framework API i serwer ASGI
4
+ fastapi==0.115.12
5
+ uvicorn==0.34.3
6
+ starlette==0.46.2
7
+
8
+ # Walidacja danych i modele
9
+ pydantic==2.11.7
10
+ pydantic_core==2.33.2
11
+ annotated-types==0.7.0
12
+ typing-extensions>=4.12.2
13
+
14
+ # Cache
15
+ fastapi-cache2==0.2.2
16
+
17
+ # NLP i uczenie maszynowe
18
+ transformers==4.44.2
19
+ tokenizers==0.19.1
20
+ torch==2.3.1+cpu --find-links https://download.pytorch.org/whl/cpu
21
+
22
+ # Przetwarzanie RSS i HTML
23
+ feedparser==6.0.11
24
+ beautifulsoup4==4.12.3
25
+ requests==2.31.0
26
+
27
+ # Konfiguracja środowiska
28
+ python-dotenv==1.0.1
29
+
30
+ # Claude API (Anthropic)
31
+ anthropic==0.40.0