Spaces:
Sleeping
Sleeping
Commit ·
95f9219
0
Parent(s):
feat: initial HF Spaces deployment
Browse files- .gitignore +10 -0
- Dockerfile +7 -0
- app/__init__.py +0 -0
- app/benchmark.py +240 -0
- app/config.py +61 -0
- app/main.py +55 -0
- app/models.py +16 -0
- app/nlp_service.py +433 -0
- app/routes/__init__.py +0 -0
- app/routes/compare.py +372 -0
- app/routes/misc.py +68 -0
- app/routes/news.py +266 -0
- app/routes/saved.py +34 -0
- app/rss_utils.py +70 -0
- app/storage.py +60 -0
- requirements.txt +31 -0
.gitignore
ADDED
|
@@ -0,0 +1,10 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
.env
|
| 2 |
+
__pycache__/
|
| 3 |
+
*.pyc
|
| 4 |
+
*.pyo
|
| 5 |
+
saved_articles.json
|
| 6 |
+
benchmark_results/
|
| 7 |
+
debug_adapters.py
|
| 8 |
+
truthscan_report.*
|
| 9 |
+
truthscan_test.py
|
| 10 |
+
setup_test_env.bat
|
Dockerfile
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
FROM python:3.11-slim
|
| 2 |
+
WORKDIR /app
|
| 3 |
+
COPY requirements.txt .
|
| 4 |
+
RUN pip install --no-cache-dir -r requirements.txt
|
| 5 |
+
COPY . .
|
| 6 |
+
EXPOSE 7860
|
| 7 |
+
CMD ["uvicorn", "app.main:app", "--host", "0.0.0.0", "--port", "7860"]
|
app/__init__.py
ADDED
|
File without changes
|
app/benchmark.py
ADDED
|
@@ -0,0 +1,240 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Moduł benchmarkowania modeli NLP.
|
| 3 |
+
|
| 4 |
+
Mierzy czas inferencji, rozkład sentymentów i prawdopodobieństwo fake news
|
| 5 |
+
dla każdego adaptera (roberta, xlm-roberta, norbert) na próbce tekstów
|
| 6 |
+
z trzech grup językowych (en, pl, no).
|
| 7 |
+
|
| 8 |
+
Użycie standalone:
|
| 9 |
+
python -m app.benchmark
|
| 10 |
+
|
| 11 |
+
Użycie z API:
|
| 12 |
+
GET /benchmark
|
| 13 |
+
GET /benchmark?adapters=roberta,xlm-roberta&langs=en,pl
|
| 14 |
+
"""
|
| 15 |
+
|
| 16 |
+
import csv
|
| 17 |
+
import json
|
| 18 |
+
import time
|
| 19 |
+
from collections import Counter
|
| 20 |
+
from pathlib import Path
|
| 21 |
+
from typing import Dict, List, Optional
|
| 22 |
+
|
| 23 |
+
from .nlp_service import get_adapter, ModelAdapter, analyze_news
|
| 24 |
+
|
| 25 |
+
# ---------------------------------------------------------------------------
|
| 26 |
+
# Próbka tekstów testowych
|
| 27 |
+
# ---------------------------------------------------------------------------
|
| 28 |
+
|
| 29 |
+
SAMPLE_TEXTS: Dict[str, List[str]] = {
|
| 30 |
+
"en": [
|
| 31 |
+
"The government announced new economic reforms to boost growth and reduce unemployment.",
|
| 32 |
+
"Flooding devastated coastal towns overnight, leaving thousands homeless.",
|
| 33 |
+
"Scientists discover a new vaccine that shows 95% efficacy against the virus.",
|
| 34 |
+
"Stock markets surged to record highs after positive inflation data.",
|
| 35 |
+
"A major scandal erupted as leaked documents exposed corporate corruption.",
|
| 36 |
+
"The peace talks collapsed after both sides failed to reach an agreement.",
|
| 37 |
+
"Renewable energy investments hit an all-time high this quarter.",
|
| 38 |
+
"Crime rates in the capital have dropped significantly over the past year.",
|
| 39 |
+
],
|
| 40 |
+
"pl": [
|
| 41 |
+
"Rząd ogłosił nowe reformy gospodarcze mające na celu pobudzenie wzrostu.",
|
| 42 |
+
"Powódź zniszczyła nadmorskie miejscowości, tysiące osób zostało bez dachu.",
|
| 43 |
+
"Naukowcy odkryli szczepionkę o 95-procentowej skuteczności przeciw wirusowi.",
|
| 44 |
+
"Giełda osiągnęła rekordowe poziomy po pozytywnych danych o inflacji.",
|
| 45 |
+
"Wybuchł wielki skandal po ujawnieniu dokumentów o korupcji korporacyjnej.",
|
| 46 |
+
"Rozmowy pokojowe załamały się po niepowodzeniu negocjacji.",
|
| 47 |
+
"Inwestycje w energię odnawialną osiągnęły historyczny rekord w tym kwartale.",
|
| 48 |
+
"Wskaźniki przestępczości w stolicy znacząco spadły w ciągu ostatniego roku.",
|
| 49 |
+
],
|
| 50 |
+
"no": [
|
| 51 |
+
"Regjeringen kunngjorde nye økonomiske reformer for å øke veksten.",
|
| 52 |
+
"Flom ødela kystbyer over natten og etterlot tusenvis uten hjem.",
|
| 53 |
+
"Forskere oppdaget en vaksine med 95 prosent effektivitet mot viruset.",
|
| 54 |
+
"Aksjemarkedene steg til rekordhøyder etter positive inflasjonsdata.",
|
| 55 |
+
"En stor skandale brøt ut da lekkede dokumenter avslørte korrupsjon.",
|
| 56 |
+
"Fredssamtalene brøt sammen etter at begge sider ikke klarte å bli enige.",
|
| 57 |
+
"Investeringer i fornybar energi nådde en historisk topp dette kvartalet.",
|
| 58 |
+
"Kriminalitetsratene i hovedstaden har falt betydelig det siste året.",
|
| 59 |
+
],
|
| 60 |
+
}
|
| 61 |
+
|
| 62 |
+
# ---------------------------------------------------------------------------
|
| 63 |
+
# Typy wyników
|
| 64 |
+
# ---------------------------------------------------------------------------
|
| 65 |
+
|
| 66 |
+
BenchmarkResult = Dict # TypedDict zastąpiony zwykłym Dict dla czytelności
|
| 67 |
+
|
| 68 |
+
# ---------------------------------------------------------------------------
|
| 69 |
+
# Funkcje benchmarkowania
|
| 70 |
+
# ---------------------------------------------------------------------------
|
| 71 |
+
|
| 72 |
+
def _run_single(
|
| 73 |
+
adapter: ModelAdapter,
|
| 74 |
+
text: str,
|
| 75 |
+
lang: str,
|
| 76 |
+
) -> Dict:
|
| 77 |
+
"""Uruchamia analyze_news dla jednego tekstu i mierzy czas."""
|
| 78 |
+
start = time.perf_counter()
|
| 79 |
+
result = analyze_news(text, lang=lang, adapter=adapter)
|
| 80 |
+
elapsed_ms = (time.perf_counter() - start) * 1000
|
| 81 |
+
return {**result, "inference_time_ms": elapsed_ms}
|
| 82 |
+
|
| 83 |
+
|
| 84 |
+
def run_benchmark(
|
| 85 |
+
adapter_names: Optional[List[str]] = None,
|
| 86 |
+
langs: Optional[List[str]] = None,
|
| 87 |
+
) -> List[BenchmarkResult]:
|
| 88 |
+
"""
|
| 89 |
+
Uruchamia benchmark dla wskazanych adapterów i języków.
|
| 90 |
+
|
| 91 |
+
Args:
|
| 92 |
+
adapter_names: Lista nazw adapterów; None = wszystkie trzy.
|
| 93 |
+
langs: Lista kodów języków; None = ['en', 'pl', 'no'].
|
| 94 |
+
|
| 95 |
+
Returns:
|
| 96 |
+
Lista słowników z wynikami — jeden wpis na kombinację adapter × język.
|
| 97 |
+
"""
|
| 98 |
+
if adapter_names is None:
|
| 99 |
+
adapter_names = ["roberta", "xlm-roberta", "norbert"]
|
| 100 |
+
if langs is None:
|
| 101 |
+
langs = ["en", "pl", "no"]
|
| 102 |
+
|
| 103 |
+
results: List[BenchmarkResult] = []
|
| 104 |
+
|
| 105 |
+
for adapter_name in adapter_names:
|
| 106 |
+
try:
|
| 107 |
+
adapter = get_adapter(adapter_name)
|
| 108 |
+
except ValueError as exc:
|
| 109 |
+
results.append({
|
| 110 |
+
"adapter_name": adapter_name,
|
| 111 |
+
"error": str(exc),
|
| 112 |
+
})
|
| 113 |
+
continue
|
| 114 |
+
|
| 115 |
+
for lang in langs:
|
| 116 |
+
texts = SAMPLE_TEXTS.get(lang, [])
|
| 117 |
+
if not texts:
|
| 118 |
+
continue
|
| 119 |
+
|
| 120 |
+
per_text: List[Dict] = []
|
| 121 |
+
for text in texts:
|
| 122 |
+
try:
|
| 123 |
+
per_text.append(_run_single(adapter, text, lang))
|
| 124 |
+
except Exception as exc:
|
| 125 |
+
per_text.append({
|
| 126 |
+
"sentiment": None,
|
| 127 |
+
"fake_probability": None,
|
| 128 |
+
"sentiment_score": None,
|
| 129 |
+
"inference_time_ms": None,
|
| 130 |
+
"error": str(exc),
|
| 131 |
+
})
|
| 132 |
+
|
| 133 |
+
# Agregacja
|
| 134 |
+
valid = [r for r in per_text if r.get("inference_time_ms") is not None]
|
| 135 |
+
times = [r["inference_time_ms"] for r in valid]
|
| 136 |
+
fakes = [r["fake_probability"] for r in valid if r.get("fake_probability") is not None]
|
| 137 |
+
sentiments = [r["sentiment"] for r in valid if r.get("sentiment")]
|
| 138 |
+
|
| 139 |
+
results.append({
|
| 140 |
+
"adapter_name": adapter_name,
|
| 141 |
+
"language": lang,
|
| 142 |
+
"sample_size": len(texts),
|
| 143 |
+
"successful_runs": len(valid),
|
| 144 |
+
"avg_inference_time_ms": round(sum(times) / len(times), 2) if times else None,
|
| 145 |
+
"min_inference_time_ms": round(min(times), 2) if times else None,
|
| 146 |
+
"max_inference_time_ms": round(max(times), 2) if times else None,
|
| 147 |
+
"avg_fake_probability": round(sum(fakes) / len(fakes), 2) if fakes else None,
|
| 148 |
+
"sentiments_distribution": dict(Counter(sentiments)),
|
| 149 |
+
"per_text": per_text,
|
| 150 |
+
})
|
| 151 |
+
|
| 152 |
+
return results
|
| 153 |
+
|
| 154 |
+
|
| 155 |
+
# ---------------------------------------------------------------------------
|
| 156 |
+
# Eksport wyników
|
| 157 |
+
# ---------------------------------------------------------------------------
|
| 158 |
+
|
| 159 |
+
def export_json(results: List[BenchmarkResult], path: Path) -> None:
|
| 160 |
+
"""Zapisuje pełne wyniki (z per_text) do pliku JSON."""
|
| 161 |
+
path.parent.mkdir(parents=True, exist_ok=True)
|
| 162 |
+
with open(path, "w", encoding="utf-8") as fh:
|
| 163 |
+
json.dump(results, fh, ensure_ascii=False, indent=2)
|
| 164 |
+
|
| 165 |
+
|
| 166 |
+
def export_csv(results: List[BenchmarkResult], path: Path) -> None:
|
| 167 |
+
"""
|
| 168 |
+
Zapisuje wyniki zbiorcze (bez per_text) do pliku CSV.
|
| 169 |
+
Jeden wiersz = jedna kombinacja adapter × język.
|
| 170 |
+
"""
|
| 171 |
+
path.parent.mkdir(parents=True, exist_ok=True)
|
| 172 |
+
summary_fields = [
|
| 173 |
+
"adapter_name", "language", "sample_size", "successful_runs",
|
| 174 |
+
"avg_inference_time_ms", "min_inference_time_ms", "max_inference_time_ms",
|
| 175 |
+
"avg_fake_probability", "sentiments_distribution",
|
| 176 |
+
]
|
| 177 |
+
with open(path, "w", newline="", encoding="utf-8") as fh:
|
| 178 |
+
writer = csv.DictWriter(fh, fieldnames=summary_fields, extrasaction="ignore")
|
| 179 |
+
writer.writeheader()
|
| 180 |
+
for row in results:
|
| 181 |
+
if "error" in row:
|
| 182 |
+
continue
|
| 183 |
+
flat = {k: row.get(k) for k in summary_fields}
|
| 184 |
+
# Rozkład sentymentów jako string JSON w komórce CSV
|
| 185 |
+
flat["sentiments_distribution"] = json.dumps(
|
| 186 |
+
row.get("sentiments_distribution", {}), ensure_ascii=False
|
| 187 |
+
)
|
| 188 |
+
writer.writerow(flat)
|
| 189 |
+
|
| 190 |
+
|
| 191 |
+
def _summary_only(results: List[BenchmarkResult]) -> List[BenchmarkResult]:
|
| 192 |
+
"""Zwraca wyniki bez pola per_text (lżejsza odpowiedź HTTP)."""
|
| 193 |
+
return [{k: v for k, v in r.items() if k != "per_text"} for r in results]
|
| 194 |
+
|
| 195 |
+
|
| 196 |
+
# ---------------------------------------------------------------------------
|
| 197 |
+
# Uruchomienie standalone
|
| 198 |
+
# ---------------------------------------------------------------------------
|
| 199 |
+
|
| 200 |
+
if __name__ == "__main__":
|
| 201 |
+
import argparse
|
| 202 |
+
|
| 203 |
+
parser = argparse.ArgumentParser(description="TruthScan NLP benchmark")
|
| 204 |
+
parser.add_argument(
|
| 205 |
+
"--adapters", default="roberta,xlm-roberta,norbert",
|
| 206 |
+
help="Przecinkowa lista adapterów (domyślnie: wszystkie)",
|
| 207 |
+
)
|
| 208 |
+
parser.add_argument(
|
| 209 |
+
"--langs", default="en,pl,no",
|
| 210 |
+
help="Przecinkowa lista języków (domyślnie: en,pl,no)",
|
| 211 |
+
)
|
| 212 |
+
parser.add_argument(
|
| 213 |
+
"--out-dir", default="benchmark_results",
|
| 214 |
+
help="Katalog wyjściowy dla plików JSON i CSV",
|
| 215 |
+
)
|
| 216 |
+
args = parser.parse_args()
|
| 217 |
+
|
| 218 |
+
adapter_names = [a.strip() for a in args.adapters.split(",")]
|
| 219 |
+
langs = [l.strip() for l in args.langs.split(",")]
|
| 220 |
+
out_dir = Path(args.out_dir)
|
| 221 |
+
|
| 222 |
+
print(f"Uruchamiam benchmark: adaptery={adapter_names}, języki={langs}")
|
| 223 |
+
results = run_benchmark(adapter_names=adapter_names, langs=langs)
|
| 224 |
+
|
| 225 |
+
json_path = out_dir / "benchmark.json"
|
| 226 |
+
csv_path = out_dir / "benchmark.csv"
|
| 227 |
+
export_json(results, json_path)
|
| 228 |
+
export_csv(results, csv_path)
|
| 229 |
+
|
| 230 |
+
print(f"Wyniki zapisane: {json_path}, {csv_path}")
|
| 231 |
+
for r in _summary_only(results):
|
| 232 |
+
if "error" in r:
|
| 233 |
+
print(f" [{r['adapter_name']}] BŁĄD: {r['error']}")
|
| 234 |
+
else:
|
| 235 |
+
print(
|
| 236 |
+
f" [{r['adapter_name']:12s} / {r['language']}] "
|
| 237 |
+
f"avg={r['avg_inference_time_ms']} ms "
|
| 238 |
+
f"fake={r['avg_fake_probability']}% "
|
| 239 |
+
f"sentiments={r['sentiments_distribution']}"
|
| 240 |
+
)
|
app/config.py
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Centralna konfiguracja aplikacji oraz stałe wykorzystywane w wielu modułach.
|
| 3 |
+
"""
|
| 4 |
+
|
| 5 |
+
import os
|
| 6 |
+
from pathlib import Path
|
| 7 |
+
|
| 8 |
+
try:
|
| 9 |
+
from dotenv import load_dotenv
|
| 10 |
+
load_dotenv(Path(__file__).resolve().parent.parent / ".env")
|
| 11 |
+
except ImportError:
|
| 12 |
+
pass # python-dotenv opcjonalne; zmienne można też ustawić w środowisku
|
| 13 |
+
|
| 14 |
+
# Konfiguracja CORS
|
| 15 |
+
# Uwaga: credentials=True jest niekompatybilne z origins=["*"] w przeglądarkach
|
| 16 |
+
# (spec CORS odrzuca tę kombinację). Dla dev/thesis używamy credentials=False.
|
| 17 |
+
CORS_ALLOW_ORIGINS = ["*"]
|
| 18 |
+
CORS_ALLOW_CREDENTIALS = False
|
| 19 |
+
CORS_ALLOW_METHODS = ["GET", "POST", "DELETE", "OPTIONS"]
|
| 20 |
+
CORS_ALLOW_HEADERS = ["*"]
|
| 21 |
+
|
| 22 |
+
# Lista obsługiwanych źródeł RSS
|
| 23 |
+
NEWS_FEEDS = {
|
| 24 |
+
# Angielskie
|
| 25 |
+
"BBC": "https://feeds.bbci.co.uk/news/rss.xml",
|
| 26 |
+
"CNN": "http://rss.cnn.com/rss/edition.rss",
|
| 27 |
+
"NYTimes": "https://rss.nytimes.com/services/xml/rss/nyt/HomePage.xml",
|
| 28 |
+
"Guardian": "https://www.theguardian.com/world/rss",
|
| 29 |
+
"AlJazeera": "https://www.aljazeera.com/xml/rss/all.xml",
|
| 30 |
+
# Polskie
|
| 31 |
+
"Money": "https://www.money.pl/rss/",
|
| 32 |
+
"PolsatNews": "https://www.polsatnews.pl/rss/wszystkie.xml",
|
| 33 |
+
"TVN24": "https://tvn24.pl/najnowsze.xml",
|
| 34 |
+
"SpidersWeb": "https://spidersweb.pl/feed",
|
| 35 |
+
"Bankier": "https://www.bankier.pl/rss/wiadomosci.xml",
|
| 36 |
+
# Norweskie
|
| 37 |
+
"NRK": "https://www.nrk.no/toppsaker.rss",
|
| 38 |
+
"VG": "https://www.vg.no/rss/feed/?limit=10",
|
| 39 |
+
"E24": "https://e24.no/rss",
|
| 40 |
+
"Aftenposten": "https://www.aftenposten.no/rss",
|
| 41 |
+
}
|
| 42 |
+
|
| 43 |
+
# Mapowanie wyników analizy sentymentu na etykiety językowe
|
| 44 |
+
SENTIMENT_MAP = {
|
| 45 |
+
"negative": {"pl": "Negatywne", "en": "Negative", "no": "Negativt"},
|
| 46 |
+
"neutral": {"pl": "Neutralne", "en": "Neutral", "no": "Nøytralt"},
|
| 47 |
+
"positive": {"pl": "Pozytywne", "en": "Positive", "no": "Positivt"},
|
| 48 |
+
}
|
| 49 |
+
|
| 50 |
+
# Ścieżka do pliku z zapisanymi artykułami
|
| 51 |
+
SAVED_FILE = Path("saved_articles.json")
|
| 52 |
+
|
| 53 |
+
# Czas życia cache (sekundy)
|
| 54 |
+
CACHE_TTL_SECONDS = 120
|
| 55 |
+
|
| 56 |
+
# Konfiguracja Redis (jeśli używany jako backend cache)
|
| 57 |
+
REDIS_URL = os.getenv("REDIS_URL", "redis://localhost:6379")
|
| 58 |
+
CACHE_TTL = 300
|
| 59 |
+
|
| 60 |
+
# Klucz API Anthropic (Claude)
|
| 61 |
+
ANTHROPIC_API_KEY = os.getenv("ANTHROPIC_API_KEY")
|
app/main.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Główna konfiguracja i uruchomienie aplikacji FastAPI.
|
| 3 |
+
"""
|
| 4 |
+
|
| 5 |
+
from fastapi import FastAPI, Request
|
| 6 |
+
from fastapi.responses import Response
|
| 7 |
+
from fastapi_cache import FastAPICache
|
| 8 |
+
from fastapi_cache.backends.inmemory import InMemoryBackend
|
| 9 |
+
|
| 10 |
+
from .routes import misc, news, saved, compare
|
| 11 |
+
|
| 12 |
+
app = FastAPI()
|
| 13 |
+
|
| 14 |
+
_CORS_HEADERS = {
|
| 15 |
+
"Access-Control-Allow-Origin": "*",
|
| 16 |
+
"Access-Control-Allow-Methods": "GET, POST, DELETE, OPTIONS",
|
| 17 |
+
"Access-Control-Allow-Headers": "*",
|
| 18 |
+
"Access-Control-Max-Age": "86400",
|
| 19 |
+
}
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
@app.middleware("http")
|
| 23 |
+
async def add_cors_headers(request: Request, call_next):
|
| 24 |
+
# Odpowiedź na preflight OPTIONS bez przekazywania do routera
|
| 25 |
+
if request.method == "OPTIONS":
|
| 26 |
+
return Response(status_code=204, headers=_CORS_HEADERS)
|
| 27 |
+
|
| 28 |
+
response = await call_next(request)
|
| 29 |
+
|
| 30 |
+
for key, value in _CORS_HEADERS.items():
|
| 31 |
+
response.headers[key] = value
|
| 32 |
+
|
| 33 |
+
return response
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
@app.on_event("startup")
|
| 37 |
+
async def startup():
|
| 38 |
+
FastAPICache.init(InMemoryBackend(), prefix="news-cache")
|
| 39 |
+
print("✓ Cache initialized (5 minut TTL)")
|
| 40 |
+
|
| 41 |
+
|
| 42 |
+
app.include_router(misc.router)
|
| 43 |
+
app.include_router(news.router)
|
| 44 |
+
app.include_router(saved.router)
|
| 45 |
+
app.include_router(compare.router)
|
| 46 |
+
|
| 47 |
+
|
| 48 |
+
@app.get("/")
|
| 49 |
+
async def root():
|
| 50 |
+
return {"message": "TruthScan API", "status": "running", "cached": True}
|
| 51 |
+
|
| 52 |
+
|
| 53 |
+
@app.get("/health")
|
| 54 |
+
async def health_check():
|
| 55 |
+
return {"status": "healthy"}
|
app/models.py
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Modele danych wykorzystywane do walidacji i serializacji artykułów.
|
| 3 |
+
"""
|
| 4 |
+
|
| 5 |
+
from pydantic import BaseModel
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
class Article(BaseModel):
|
| 9 |
+
# Model artykułu wykorzystywany w komunikacji API
|
| 10 |
+
title: str
|
| 11 |
+
link: str
|
| 12 |
+
summary: str
|
| 13 |
+
published: str
|
| 14 |
+
sentiment: str
|
| 15 |
+
fake_probability: float
|
| 16 |
+
source: str
|
app/nlp_service.py
ADDED
|
@@ -0,0 +1,433 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Serwis NLP z architekturą plug-in.
|
| 3 |
+
|
| 4 |
+
Każdy model jest reprezentowany przez adapter dziedziczący z ModelAdapter.
|
| 5 |
+
Publiczne API (analyze_news, analyze_news_batch) pozostaje niezmienione,
|
| 6 |
+
więc routes/news.py nie wymaga modyfikacji.
|
| 7 |
+
|
| 8 |
+
Fake news detection używa jednej globalnej instancji BART (_shared_bart)
|
| 9 |
+
współdzielonej przez wszystkie adaptery — model ładowany jest leniwie
|
| 10 |
+
przy pierwszym wywołaniu get_shared_bart().
|
| 11 |
+
"""
|
| 12 |
+
|
| 13 |
+
from abc import ABC, abstractmethod
|
| 14 |
+
from typing import Dict, Any, List, Optional
|
| 15 |
+
from concurrent.futures import ThreadPoolExecutor
|
| 16 |
+
|
| 17 |
+
from transformers import pipeline
|
| 18 |
+
|
| 19 |
+
from .config import SENTIMENT_MAP
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
# ---------------------------------------------------------------------------
|
| 23 |
+
# Współdzielony BART do wykrywania fake news (jeden egzemplarz dla wszystkich)
|
| 24 |
+
# ---------------------------------------------------------------------------
|
| 25 |
+
|
| 26 |
+
_shared_bart = None
|
| 27 |
+
|
| 28 |
+
|
| 29 |
+
def get_shared_bart():
|
| 30 |
+
"""
|
| 31 |
+
Zwraca globalną instancję potoku zero-shot-classification (BART).
|
| 32 |
+
Ładuje model przy pierwszym wywołaniu (lazy init).
|
| 33 |
+
"""
|
| 34 |
+
global _shared_bart
|
| 35 |
+
if _shared_bart is None:
|
| 36 |
+
_shared_bart = pipeline(
|
| 37 |
+
"zero-shot-classification",
|
| 38 |
+
model="facebook/bart-large-mnli",
|
| 39 |
+
)
|
| 40 |
+
return _shared_bart
|
| 41 |
+
|
| 42 |
+
|
| 43 |
+
def analyze_fake_news_shared(text: str) -> float:
|
| 44 |
+
"""
|
| 45 |
+
Klasyfikuje tekst jako real/fake używając współdzielonego BART.
|
| 46 |
+
|
| 47 |
+
Zwraca:
|
| 48 |
+
Prawdopodobieństwo fake news w procentach (0.0 – 100.0).
|
| 49 |
+
"""
|
| 50 |
+
result = get_shared_bart()(text, candidate_labels=["real", "fake"])
|
| 51 |
+
for lbl, score in zip(result["labels"], result["scores"]):
|
| 52 |
+
if lbl == "fake":
|
| 53 |
+
return round(score * 100, 2)
|
| 54 |
+
return 0.0
|
| 55 |
+
|
| 56 |
+
|
| 57 |
+
# ---------------------------------------------------------------------------
|
| 58 |
+
# Klasa bazowa
|
| 59 |
+
# ---------------------------------------------------------------------------
|
| 60 |
+
|
| 61 |
+
class ModelAdapter(ABC):
|
| 62 |
+
"""
|
| 63 |
+
Abstrakcyjny adapter modelu NLP.
|
| 64 |
+
|
| 65 |
+
Każdy konkretny adapter implementuje:
|
| 66 |
+
- analyze_sentiment(text) -> {"label": str, "score": float}
|
| 67 |
+
- analyze_fake_news(text) -> {"labels": List[str], "scores": List[float]}
|
| 68 |
+
|
| 69 |
+
analyze_fake_news deleguje do współdzielonego BART (get_shared_bart),
|
| 70 |
+
więc każda podklasa może użyć domyślnej implementacji z tej klasy bazowej.
|
| 71 |
+
"""
|
| 72 |
+
|
| 73 |
+
@property
|
| 74 |
+
@abstractmethod
|
| 75 |
+
def name(self) -> str:
|
| 76 |
+
...
|
| 77 |
+
|
| 78 |
+
@property
|
| 79 |
+
@abstractmethod
|
| 80 |
+
def supported_languages(self) -> List[str]:
|
| 81 |
+
...
|
| 82 |
+
|
| 83 |
+
@abstractmethod
|
| 84 |
+
def analyze_sentiment(self, text: str) -> Dict[str, Any]:
|
| 85 |
+
"""
|
| 86 |
+
Zwraca {"label": str, "score": float}
|
| 87 |
+
gdzie label to: 'positive' | 'negative' | 'neutral'
|
| 88 |
+
"""
|
| 89 |
+
...
|
| 90 |
+
|
| 91 |
+
def analyze_fake_news(self, text: str) -> Dict[str, Any]:
|
| 92 |
+
"""
|
| 93 |
+
Klasyfikuje tekst używając współdzielonego BART.
|
| 94 |
+
Podklasy mogą nadpisać, ale domyślnie korzystają z _shared_bart.
|
| 95 |
+
"""
|
| 96 |
+
result = get_shared_bart()(text, candidate_labels=["real", "fake"])
|
| 97 |
+
return {"labels": result["labels"], "scores": result["scores"]}
|
| 98 |
+
|
| 99 |
+
|
| 100 |
+
# ---------------------------------------------------------------------------
|
| 101 |
+
# Helpers
|
| 102 |
+
# ---------------------------------------------------------------------------
|
| 103 |
+
|
| 104 |
+
def _extract_pipeline_result(adapter_name: str, raw: Any) -> Dict[str, Any]:
|
| 105 |
+
"""
|
| 106 |
+
Bezpiecznie wyciąga słownik {label, score} z surowego outputu pipeline.
|
| 107 |
+
|
| 108 |
+
Pipeline text-classification może zwrócić:
|
| 109 |
+
- [{"label": ..., "score": ...}] (return_all_scores=False)
|
| 110 |
+
- [[{"label": ..., "score": ...}, ...]] (return_all_scores=True / top_k=None)
|
| 111 |
+
"""
|
| 112 |
+
print(f"[DEBUG] {adapter_name}: raw = {raw!r}")
|
| 113 |
+
try:
|
| 114 |
+
item = raw[0]
|
| 115 |
+
except (IndexError, TypeError) as exc:
|
| 116 |
+
print(f"[DEBUG] {adapter_name}: raw[0] failed: {exc}")
|
| 117 |
+
return {"label": "", "score": 0.0}
|
| 118 |
+
|
| 119 |
+
if isinstance(item, dict):
|
| 120 |
+
result = item
|
| 121 |
+
elif isinstance(item, list):
|
| 122 |
+
# return_all_scores=True zwraca listę list — bierzemy element o najwyższym score
|
| 123 |
+
if not item:
|
| 124 |
+
print(f"[DEBUG] {adapter_name}: item jest pustą listą")
|
| 125 |
+
return {"label": "", "score": 0.0}
|
| 126 |
+
result = max(item, key=lambda x: x.get("score", 0.0))
|
| 127 |
+
else:
|
| 128 |
+
print(f"[DEBUG] {adapter_name}: nieoczekiwany typ item: {type(item)!r}, raw={raw!r}")
|
| 129 |
+
return {"label": "", "score": 0.0}
|
| 130 |
+
|
| 131 |
+
print(f"[DEBUG] {adapter_name}: extracted = {result!r}")
|
| 132 |
+
return result
|
| 133 |
+
|
| 134 |
+
|
| 135 |
+
def _normalize_sentiment_label(raw_label: str) -> str:
|
| 136 |
+
label = (raw_label or "").strip().lower()
|
| 137 |
+
|
| 138 |
+
_POSITIVE = {"positive", "pos", "label_2", "2", "very positive"}
|
| 139 |
+
_NEGATIVE = {"negative", "neg", "label_0", "0", "very negative"}
|
| 140 |
+
# "mixed" (NorBERT3 sentence-sentiment) → traktujemy jako neutral
|
| 141 |
+
_NEUTRAL = {"neutral", "mixed", "label_1"}
|
| 142 |
+
|
| 143 |
+
if label in _POSITIVE or label.startswith("pos"):
|
| 144 |
+
return "positive"
|
| 145 |
+
if label in _NEGATIVE or label.startswith("neg"):
|
| 146 |
+
return "negative"
|
| 147 |
+
if label in _NEUTRAL:
|
| 148 |
+
return "neutral"
|
| 149 |
+
return "neutral"
|
| 150 |
+
|
| 151 |
+
|
| 152 |
+
# ---------------------------------------------------------------------------
|
| 153 |
+
# Adaptery — każdy ładuje tylko swój model sentymentu
|
| 154 |
+
# ---------------------------------------------------------------------------
|
| 155 |
+
|
| 156 |
+
class RoBERTaAdapter(ModelAdapter):
|
| 157 |
+
"""
|
| 158 |
+
Adapter anglojęzyczny.
|
| 159 |
+
sentyment: cardiffnlp/twitter-roberta-base-sentiment-latest
|
| 160 |
+
fake news: współdzielony BART (ModelAdapter.analyze_fake_news)
|
| 161 |
+
"""
|
| 162 |
+
|
| 163 |
+
def __init__(self) -> None:
|
| 164 |
+
self._sentiment_pipe = None
|
| 165 |
+
|
| 166 |
+
def _load(self) -> None:
|
| 167 |
+
if self._sentiment_pipe is None:
|
| 168 |
+
self._sentiment_pipe = pipeline(
|
| 169 |
+
"text-classification",
|
| 170 |
+
model="cardiffnlp/twitter-roberta-base-sentiment-latest",
|
| 171 |
+
return_all_scores=False,
|
| 172 |
+
)
|
| 173 |
+
|
| 174 |
+
@property
|
| 175 |
+
def name(self) -> str:
|
| 176 |
+
return "roberta"
|
| 177 |
+
|
| 178 |
+
@property
|
| 179 |
+
def supported_languages(self) -> List[str]:
|
| 180 |
+
return ["en"]
|
| 181 |
+
|
| 182 |
+
def analyze_sentiment(self, text: str) -> Dict[str, Any]:
|
| 183 |
+
self._load()
|
| 184 |
+
raw = self._sentiment_pipe(text)
|
| 185 |
+
result = _extract_pipeline_result(self.name, raw)
|
| 186 |
+
return {
|
| 187 |
+
"label": _normalize_sentiment_label(result.get("label", "")),
|
| 188 |
+
"score": float(result.get("score", 0.0)),
|
| 189 |
+
}
|
| 190 |
+
|
| 191 |
+
|
| 192 |
+
class XLMRoBERTaAdapter(ModelAdapter):
|
| 193 |
+
"""
|
| 194 |
+
Adapter wielojęzyczny (pl, en, no).
|
| 195 |
+
sentyment: cardiffnlp/twitter-xlm-roberta-base-sentiment
|
| 196 |
+
fake news: współdzielony BART
|
| 197 |
+
"""
|
| 198 |
+
|
| 199 |
+
def __init__(self) -> None:
|
| 200 |
+
self._sentiment_pipe = None
|
| 201 |
+
|
| 202 |
+
def _load(self) -> None:
|
| 203 |
+
if self._sentiment_pipe is None:
|
| 204 |
+
self._sentiment_pipe = pipeline(
|
| 205 |
+
"text-classification",
|
| 206 |
+
model="cardiffnlp/twitter-xlm-roberta-base-sentiment",
|
| 207 |
+
return_all_scores=False,
|
| 208 |
+
)
|
| 209 |
+
|
| 210 |
+
@property
|
| 211 |
+
def name(self) -> str:
|
| 212 |
+
return "xlm-roberta"
|
| 213 |
+
|
| 214 |
+
@property
|
| 215 |
+
def supported_languages(self) -> List[str]:
|
| 216 |
+
return ["pl", "en", "no"]
|
| 217 |
+
|
| 218 |
+
def analyze_sentiment(self, text: str) -> Dict[str, Any]:
|
| 219 |
+
self._load()
|
| 220 |
+
raw = self._sentiment_pipe(text)
|
| 221 |
+
result = _extract_pipeline_result(self.name, raw)
|
| 222 |
+
return {
|
| 223 |
+
"label": _normalize_sentiment_label(result.get("label", "")),
|
| 224 |
+
"score": float(result.get("score", 0.0)),
|
| 225 |
+
}
|
| 226 |
+
|
| 227 |
+
|
| 228 |
+
class HerBERTAdapter(ModelAdapter):
|
| 229 |
+
"""
|
| 230 |
+
Adapter dla języka polskiego (HerBERT).
|
| 231 |
+
sentyment: allegro/herbert-base-cased (model bazowy — wymaga fine-tuningu)
|
| 232 |
+
fake news: współdzielony BART
|
| 233 |
+
"""
|
| 234 |
+
|
| 235 |
+
SENTIMENT_MODEL: str = "allegro/herbert-base-cased"
|
| 236 |
+
|
| 237 |
+
def __init__(self, sentiment_model: Optional[str] = None) -> None:
|
| 238 |
+
self._sentiment_model_id = sentiment_model or self.SENTIMENT_MODEL
|
| 239 |
+
self._sentiment_pipe = None
|
| 240 |
+
|
| 241 |
+
def _load(self) -> None:
|
| 242 |
+
if self._sentiment_pipe is None:
|
| 243 |
+
self._sentiment_pipe = pipeline(
|
| 244 |
+
"text-classification",
|
| 245 |
+
model=self._sentiment_model_id,
|
| 246 |
+
return_all_scores=False,
|
| 247 |
+
)
|
| 248 |
+
|
| 249 |
+
@property
|
| 250 |
+
def name(self) -> str:
|
| 251 |
+
return "herbert"
|
| 252 |
+
|
| 253 |
+
@property
|
| 254 |
+
def supported_languages(self) -> List[str]:
|
| 255 |
+
return ["pl"]
|
| 256 |
+
|
| 257 |
+
def analyze_sentiment(self, text: str) -> Dict[str, Any]:
|
| 258 |
+
self._load()
|
| 259 |
+
raw = self._sentiment_pipe(text)
|
| 260 |
+
result = _extract_pipeline_result(self.name, raw)
|
| 261 |
+
return {
|
| 262 |
+
"label": _normalize_sentiment_label(result.get("label", "")),
|
| 263 |
+
"score": float(result.get("score", 0.0)),
|
| 264 |
+
}
|
| 265 |
+
|
| 266 |
+
|
| 267 |
+
class NorBERTAdapter(ModelAdapter):
|
| 268 |
+
"""
|
| 269 |
+
Adapter dla języka norweskiego.
|
| 270 |
+
sentyment: cardiffnlp/twitter-xlm-roberta-base-sentiment-multilingual
|
| 271 |
+
(wielojęzyczny XLM-RoBERTa Cardiff, obsługuje norweski)
|
| 272 |
+
fake news: współdzielony BART
|
| 273 |
+
"""
|
| 274 |
+
|
| 275 |
+
# cardiffnlp/twitter-xlm-roberta-base-sentiment-multilingual —
|
| 276 |
+
# wielojęzyczny model Cardiff fine-tuned na sentymencie (obsługuje norweski),
|
| 277 |
+
# standardowa architektura XLM-RoBERTa kompatybilna z pipeline("text-classification").
|
| 278 |
+
SENTIMENT_MODEL: str = "cardiffnlp/twitter-xlm-roberta-base-sentiment-multilingual"
|
| 279 |
+
|
| 280 |
+
def __init__(self, sentiment_model: Optional[str] = None) -> None:
|
| 281 |
+
self._sentiment_model_id = sentiment_model or self.SENTIMENT_MODEL
|
| 282 |
+
self._sentiment_pipe = None
|
| 283 |
+
|
| 284 |
+
def _load(self) -> None:
|
| 285 |
+
if self._sentiment_pipe is None:
|
| 286 |
+
self._sentiment_pipe = pipeline(
|
| 287 |
+
"text-classification",
|
| 288 |
+
model=self._sentiment_model_id,
|
| 289 |
+
return_all_scores=False,
|
| 290 |
+
)
|
| 291 |
+
|
| 292 |
+
@property
|
| 293 |
+
def name(self) -> str:
|
| 294 |
+
return "norbert"
|
| 295 |
+
|
| 296 |
+
@property
|
| 297 |
+
def supported_languages(self) -> List[str]:
|
| 298 |
+
return ["no"]
|
| 299 |
+
|
| 300 |
+
def analyze_sentiment(self, text: str) -> Dict[str, Any]:
|
| 301 |
+
self._load()
|
| 302 |
+
raw = self._sentiment_pipe(text)
|
| 303 |
+
result = _extract_pipeline_result(self.name, raw)
|
| 304 |
+
return {
|
| 305 |
+
"label": _normalize_sentiment_label(result.get("label", "")),
|
| 306 |
+
"score": float(result.get("score", 0.0)),
|
| 307 |
+
}
|
| 308 |
+
|
| 309 |
+
|
| 310 |
+
# ---------------------------------------------------------------------------
|
| 311 |
+
# Rejestr adapterów — lazy initialization
|
| 312 |
+
# ---------------------------------------------------------------------------
|
| 313 |
+
|
| 314 |
+
_REGISTRY: Dict[str, ModelAdapter] = {}
|
| 315 |
+
_active_adapter: Optional[ModelAdapter] = None
|
| 316 |
+
executor = ThreadPoolExecutor(max_workers=3)
|
| 317 |
+
_BUILTIN_ADAPTERS = {"roberta", "xlm-roberta", "herbert", "norbert"}
|
| 318 |
+
|
| 319 |
+
|
| 320 |
+
def _init_registry() -> None:
|
| 321 |
+
if _REGISTRY:
|
| 322 |
+
return
|
| 323 |
+
_REGISTRY["roberta"] = RoBERTaAdapter()
|
| 324 |
+
_REGISTRY["xlm-roberta"] = XLMRoBERTaAdapter()
|
| 325 |
+
_REGISTRY["herbert"] = HerBERTAdapter()
|
| 326 |
+
_REGISTRY["norbert"] = NorBERTAdapter()
|
| 327 |
+
|
| 328 |
+
|
| 329 |
+
def get_adapter(name: str) -> ModelAdapter:
|
| 330 |
+
_init_registry()
|
| 331 |
+
if name not in _REGISTRY:
|
| 332 |
+
raise ValueError(
|
| 333 |
+
f"Nieznany adapter: '{name}'. Dostępne: {list(_REGISTRY.keys())}"
|
| 334 |
+
)
|
| 335 |
+
return _REGISTRY[name]
|
| 336 |
+
|
| 337 |
+
|
| 338 |
+
def set_active_adapter(name: str) -> None:
|
| 339 |
+
global _active_adapter
|
| 340 |
+
_active_adapter = get_adapter(name)
|
| 341 |
+
|
| 342 |
+
|
| 343 |
+
def register_adapter(adapter: ModelAdapter) -> None:
|
| 344 |
+
_init_registry()
|
| 345 |
+
_REGISTRY[adapter.name] = adapter
|
| 346 |
+
|
| 347 |
+
|
| 348 |
+
def _get_active_adapter() -> ModelAdapter:
|
| 349 |
+
global _active_adapter
|
| 350 |
+
if _active_adapter is None:
|
| 351 |
+
_active_adapter = get_adapter("roberta")
|
| 352 |
+
return _active_adapter
|
| 353 |
+
|
| 354 |
+
|
| 355 |
+
# ---------------------------------------------------------------------------
|
| 356 |
+
# Publiczne API
|
| 357 |
+
# ---------------------------------------------------------------------------
|
| 358 |
+
|
| 359 |
+
def analyze_news(
|
| 360 |
+
text: str,
|
| 361 |
+
lang: str = "pl",
|
| 362 |
+
adapter: Optional[ModelAdapter] = None,
|
| 363 |
+
) -> dict:
|
| 364 |
+
"""
|
| 365 |
+
Analizuje pojedynczy tekst pod kątem sentymentu i fake news.
|
| 366 |
+
|
| 367 |
+
Returns:
|
| 368 |
+
{"sentiment": str, "fake_probability": float, "sentiment_score": float}
|
| 369 |
+
"""
|
| 370 |
+
_neutral = SENTIMENT_MAP["neutral"].get(lang, "Neutral")
|
| 371 |
+
|
| 372 |
+
if not text or len(text.strip()) < 10:
|
| 373 |
+
return {"sentiment": _neutral, "fake_probability": 0.0, "sentiment_score": 0.0}
|
| 374 |
+
|
| 375 |
+
_adapter = adapter or _get_active_adapter()
|
| 376 |
+
|
| 377 |
+
try:
|
| 378 |
+
sentiment_result = _adapter.analyze_sentiment(text)
|
| 379 |
+
fake_result = _adapter.analyze_fake_news(text)
|
| 380 |
+
except Exception:
|
| 381 |
+
return {"sentiment": _neutral, "fake_probability": 0.0, "sentiment_score": 0.0}
|
| 382 |
+
|
| 383 |
+
label = sentiment_result.get("label", "neutral")
|
| 384 |
+
sentiment_translated = SENTIMENT_MAP.get(label, {}).get(lang, _neutral)
|
| 385 |
+
|
| 386 |
+
fake_score = 0.0
|
| 387 |
+
for lbl, score in zip(fake_result["labels"], fake_result["scores"]):
|
| 388 |
+
if lbl == "fake":
|
| 389 |
+
fake_score = score
|
| 390 |
+
break
|
| 391 |
+
|
| 392 |
+
return {
|
| 393 |
+
"sentiment": sentiment_translated,
|
| 394 |
+
"fake_probability": round(fake_score * 100, 2),
|
| 395 |
+
"sentiment_score": round(float(sentiment_result.get("score", 0.0)), 2),
|
| 396 |
+
}
|
| 397 |
+
|
| 398 |
+
|
| 399 |
+
def analyze_news_batch(
|
| 400 |
+
texts: List[str],
|
| 401 |
+
lang: str = "pl",
|
| 402 |
+
adapter: Optional[ModelAdapter] = None,
|
| 403 |
+
) -> List[Dict[str, Any]]:
|
| 404 |
+
"""Analiza wielu tekstów w trybie batch z użyciem ThreadPoolExecutor."""
|
| 405 |
+
if not texts:
|
| 406 |
+
return []
|
| 407 |
+
|
| 408 |
+
_adapter = adapter or _get_active_adapter()
|
| 409 |
+
results: List[Dict[str, Any]] = []
|
| 410 |
+
batch_size = 3
|
| 411 |
+
|
| 412 |
+
batches = [texts[i:i + batch_size] for i in range(0, len(texts), batch_size)]
|
| 413 |
+
|
| 414 |
+
for batch in batches:
|
| 415 |
+
futures = [
|
| 416 |
+
executor.submit(analyze_news, text, lang, _adapter)
|
| 417 |
+
for text in batch
|
| 418 |
+
]
|
| 419 |
+
for future in futures:
|
| 420 |
+
try:
|
| 421 |
+
results.append(future.result())
|
| 422 |
+
except Exception:
|
| 423 |
+
_neutral = SENTIMENT_MAP["neutral"].get(lang, "Neutral")
|
| 424 |
+
results.append(
|
| 425 |
+
{"sentiment": _neutral, "fake_probability": 0.0, "sentiment_score": 0.0}
|
| 426 |
+
)
|
| 427 |
+
|
| 428 |
+
return results
|
| 429 |
+
|
| 430 |
+
|
| 431 |
+
def analyze_news_single(text: str, lang: str) -> Dict[str, Any]:
|
| 432 |
+
"""Wrapper dla analizy pojedynczego tekstu (używany w batch processing)."""
|
| 433 |
+
return analyze_news(text, lang)
|
app/routes/__init__.py
ADDED
|
File without changes
|
app/routes/compare.py
ADDED
|
@@ -0,0 +1,372 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Endpoint porównujący wyniki wszystkich adapterów NLP dla jednego tekstu.
|
| 3 |
+
|
| 4 |
+
Architektura:
|
| 5 |
+
- fake news: jeden wspólny wynik z _shared_bart (BART large MNLI)
|
| 6 |
+
- sentyment: każdy z 4 adapterów używa własnego modelu
|
| 7 |
+
|
| 8 |
+
Zwraca:
|
| 9 |
+
- shared_fake_probability: float — wynik BART (jeden dla wszystkich)
|
| 10 |
+
- results: lista wyników sentymentu per adapter z czasem inferencji
|
| 11 |
+
- explanation: wyjaśnienie różnic wygenerowane przez Claude API
|
| 12 |
+
"""
|
| 13 |
+
|
| 14 |
+
import time
|
| 15 |
+
from typing import List
|
| 16 |
+
|
| 17 |
+
import anthropic
|
| 18 |
+
from fastapi import APIRouter, HTTPException
|
| 19 |
+
from pydantic import BaseModel
|
| 20 |
+
|
| 21 |
+
from ..config import ANTHROPIC_API_KEY, SENTIMENT_MAP
|
| 22 |
+
from ..nlp_service import get_adapter, analyze_fake_news_shared
|
| 23 |
+
|
| 24 |
+
router = APIRouter()
|
| 25 |
+
|
| 26 |
+
ADAPTER_NAMES = ["roberta", "xlm-roberta", "norbert", "herbert"]
|
| 27 |
+
|
| 28 |
+
LANG_NAMES = {
|
| 29 |
+
"pl": "polski",
|
| 30 |
+
"en": "angielski",
|
| 31 |
+
"no": "norweski",
|
| 32 |
+
}
|
| 33 |
+
|
| 34 |
+
# Język odpowiedzi Claude zależny od lang w żądaniu
|
| 35 |
+
RESPONSE_LANG = {
|
| 36 |
+
"pl": "polskim",
|
| 37 |
+
"en": "English",
|
| 38 |
+
"no": "norsk",
|
| 39 |
+
}
|
| 40 |
+
|
| 41 |
+
# Instrukcja do Claude w danym języku
|
| 42 |
+
EXPLAIN_INSTRUCTION = {
|
| 43 |
+
"pl": (
|
| 44 |
+
"Wyjaśnij w 3–4 zdaniach po polsku dlaczego modele oceniają sentyment tego tekstu tak jak oceniają. "
|
| 45 |
+
"Uwzględnij: (1) czy tekst jest w języku {lang_name} i jak to wpływa na modele trenowane na innych językach, "
|
| 46 |
+
"(2) co powoduje ewentualne rozbieżności między modelami, "
|
| 47 |
+
"(3) który model jest prawdopodobnie najbardziej trafny dla tego języka i dlaczego. "
|
| 48 |
+
"Pisz czystym tekstem — bez formatowania markdown, bez gwiazdek, bez nagłówków, bez list numerowanych."
|
| 49 |
+
),
|
| 50 |
+
"en": (
|
| 51 |
+
"Explain in 3–4 sentences in English why the models assess the sentiment of this text the way they do. "
|
| 52 |
+
"Cover: (1) whether the text is in {lang_name} and how that affects models trained on other languages, "
|
| 53 |
+
"(2) what causes any disagreements between the models, "
|
| 54 |
+
"(3) which model is likely most accurate for this language and why. "
|
| 55 |
+
"Write in plain text — no markdown, no asterisks, no headings, no numbered lists."
|
| 56 |
+
),
|
| 57 |
+
"no": (
|
| 58 |
+
"Forklar på norsk i 3–4 setninger hvorfor modellene vurderer sentimentet til denne teksten slik de gjør. "
|
| 59 |
+
"Ta med: (1) om teksten er på {lang_name} og hvordan det påvirker modeller trent på andre språk, "
|
| 60 |
+
"(2) hva som forårsaker eventuelle uenigheter mellom modellene, "
|
| 61 |
+
"(3) hvilken modell som sannsynligvis er mest nøyaktig for dette språket og hvorfor. "
|
| 62 |
+
"Skriv i ren tekst — ingen markdown, ingen stjerner, ingen overskrifter, ingen nummererte lister."
|
| 63 |
+
),
|
| 64 |
+
}
|
| 65 |
+
|
| 66 |
+
ADAPTER_DESCRIPTIONS = {
|
| 67 |
+
"roberta": "RoBERTa (trenowany na angielskich tweetach)",
|
| 68 |
+
"xlm-roberta": "XLM-RoBERTa (wielojęzyczny, pl/en/no)",
|
| 69 |
+
"norbert": "NorBERT 3 (norweski model bazowy)",
|
| 70 |
+
"herbert": "HerBERT (polski model bazowy)",
|
| 71 |
+
}
|
| 72 |
+
|
| 73 |
+
|
| 74 |
+
CHART_COMMENT_PROMPT = {
|
| 75 |
+
"pl": (
|
| 76 |
+
"Na podstawie poniższych danych fake-news probability (0–100%) dla {n} źródeł mediów "
|
| 77 |
+
"napisz dokładnie 2–3 zdania komentarza analitycznego po polsku. "
|
| 78 |
+
"Wskaż źródło z najwyższym i najniższym wynikiem oraz co to może oznaczać dla czytelnika. "
|
| 79 |
+
"Bądź zwięzły i obiektywny. Pisz czystym tekstem — bez markdown, bez gwiazdek, bez list.\n\n"
|
| 80 |
+
"Dane:\n{rows}"
|
| 81 |
+
),
|
| 82 |
+
"en": (
|
| 83 |
+
"Based on the following fake-news probability data (0–100%) for {n} media sources "
|
| 84 |
+
"write exactly 2–3 sentences of analytical commentary in English. "
|
| 85 |
+
"Identify the source with the highest and lowest score and what this might mean for readers. "
|
| 86 |
+
"Be concise and objective. Write in plain text — no markdown, no asterisks, no lists.\n\n"
|
| 87 |
+
"Data:\n{rows}"
|
| 88 |
+
),
|
| 89 |
+
"no": (
|
| 90 |
+
"Basert på følgende fake-news sannsynlighetsdata (0–100 %) for {n} mediekilder "
|
| 91 |
+
"skriv nøyaktig 2–3 setninger med analytisk kommentar på norsk. "
|
| 92 |
+
"Pek på kilden med høyest og lavest score og hva dette kan bety for leseren. "
|
| 93 |
+
"Vær kortfattet og objektiv. Skriv i ren tekst — ingen markdown, ingen stjerner, ingen lister.\n\n"
|
| 94 |
+
"Data:\n{rows}"
|
| 95 |
+
),
|
| 96 |
+
}
|
| 97 |
+
|
| 98 |
+
|
| 99 |
+
class ChartCommentRequest(BaseModel):
|
| 100 |
+
bars: list # [{"label": "BBC", "value": 23.45}, ...]
|
| 101 |
+
lang: str = "pl"
|
| 102 |
+
|
| 103 |
+
|
| 104 |
+
class ChartCommentResponse(BaseModel):
|
| 105 |
+
comment: str
|
| 106 |
+
|
| 107 |
+
|
| 108 |
+
@router.post("/ai-chart-comment", response_model=ChartCommentResponse)
|
| 109 |
+
def ai_chart_comment(request: ChartCommentRequest):
|
| 110 |
+
"""Generuje krótki komentarz analityczny Claude do wykresu fake-news probability."""
|
| 111 |
+
if not request.bars:
|
| 112 |
+
raise HTTPException(status_code=422, detail="Brak danych wykresu.")
|
| 113 |
+
|
| 114 |
+
lang = request.lang if request.lang in CHART_COMMENT_PROMPT else "pl"
|
| 115 |
+
|
| 116 |
+
# Buduj czytelny opis danych (posortowany malejąco po value)
|
| 117 |
+
sorted_bars = sorted(request.bars, key=lambda b: b.get("value", 0), reverse=True)
|
| 118 |
+
rows = "\n".join(
|
| 119 |
+
f" {b['label']}: {b.get('value', 0):.2f}%"
|
| 120 |
+
for b in sorted_bars
|
| 121 |
+
if b.get("label")
|
| 122 |
+
)
|
| 123 |
+
prompt = CHART_COMMENT_PROMPT[lang].format(n=len(sorted_bars), rows=rows)
|
| 124 |
+
|
| 125 |
+
if not ANTHROPIC_API_KEY:
|
| 126 |
+
return ChartCommentResponse(comment="[Brak klucza ANTHROPIC_API_KEY]")
|
| 127 |
+
|
| 128 |
+
try:
|
| 129 |
+
client = anthropic.Anthropic(api_key=ANTHROPIC_API_KEY)
|
| 130 |
+
message = client.messages.create(
|
| 131 |
+
model="claude-sonnet-4-20250514",
|
| 132 |
+
max_tokens=300,
|
| 133 |
+
messages=[{"role": "user", "content": prompt}],
|
| 134 |
+
)
|
| 135 |
+
comment = message.content[0].text.strip() if message.content else ""
|
| 136 |
+
except Exception as e:
|
| 137 |
+
comment = f"[Błąd Claude API: {e}]"
|
| 138 |
+
|
| 139 |
+
return ChartCommentResponse(comment=comment)
|
| 140 |
+
|
| 141 |
+
|
| 142 |
+
EMOTION_COMMENT_PROMPT = {
|
| 143 |
+
"pl": (
|
| 144 |
+
"Na podstawie danych o emocjach w artykułach: "
|
| 145 |
+
"Pozytywne: {positive}, Neutralne: {neutral}, Negatywne: {negative} "
|
| 146 |
+
"(łącznie {total} artykułów) — napisz 2–3 zdania komentarza analitycznego po polsku. "
|
| 147 |
+
"Co dominujący sentyment mówi o obecnym klimacie medialnym? "
|
| 148 |
+
"Bądź zwięzły i obiektywny. Pisz czystym tekstem — bez markdown, bez gwiazdek, bez list."
|
| 149 |
+
),
|
| 150 |
+
"en": (
|
| 151 |
+
"Based on emotion data from articles: "
|
| 152 |
+
"Positive: {positive}, Neutral: {neutral}, Negative: {negative} "
|
| 153 |
+
"(total {total} articles) — write 2–3 sentences of analytical commentary in English. "
|
| 154 |
+
"What does the dominant sentiment say about the current media climate? "
|
| 155 |
+
"Be concise and objective. Write in plain text — no markdown, no asterisks, no lists."
|
| 156 |
+
),
|
| 157 |
+
"no": (
|
| 158 |
+
"Basert på emosjonelle data fra artikler: "
|
| 159 |
+
"Positive: {positive}, Nøytrale: {neutral}, Negative: {negative} "
|
| 160 |
+
"(totalt {total} artikler) — skriv 2–3 setninger med analytisk kommentar på norsk. "
|
| 161 |
+
"Hva sier det dominerende sentimentet om det nåværende mediaklimaet? "
|
| 162 |
+
"Vær kortfattet og objektiv. Skriv i ren tekst — ingen markdown, ingen stjerner, ingen lister."
|
| 163 |
+
),
|
| 164 |
+
}
|
| 165 |
+
|
| 166 |
+
|
| 167 |
+
class EmotionCommentRequest(BaseModel):
|
| 168 |
+
positive: int = 0
|
| 169 |
+
neutral: int = 0
|
| 170 |
+
negative: int = 0
|
| 171 |
+
lang: str = "pl"
|
| 172 |
+
|
| 173 |
+
|
| 174 |
+
class EmotionCommentResponse(BaseModel):
|
| 175 |
+
comment: str
|
| 176 |
+
|
| 177 |
+
|
| 178 |
+
@router.post("/ai-emotion-comment", response_model=EmotionCommentResponse)
|
| 179 |
+
def ai_emotion_comment(request: EmotionCommentRequest):
|
| 180 |
+
"""Generuje krótki komentarz analityczny Claude do wykresu emocji."""
|
| 181 |
+
total = request.positive + request.neutral + request.negative
|
| 182 |
+
if total == 0:
|
| 183 |
+
raise HTTPException(status_code=422, detail="Brak danych emocji.")
|
| 184 |
+
|
| 185 |
+
lang = request.lang if request.lang in EMOTION_COMMENT_PROMPT else "pl"
|
| 186 |
+
prompt = EMOTION_COMMENT_PROMPT[lang].format(
|
| 187 |
+
positive=request.positive,
|
| 188 |
+
neutral=request.neutral,
|
| 189 |
+
negative=request.negative,
|
| 190 |
+
total=total,
|
| 191 |
+
)
|
| 192 |
+
|
| 193 |
+
if not ANTHROPIC_API_KEY:
|
| 194 |
+
return EmotionCommentResponse(comment="[Brak klucza ANTHROPIC_API_KEY]")
|
| 195 |
+
|
| 196 |
+
try:
|
| 197 |
+
client = anthropic.Anthropic(api_key=ANTHROPIC_API_KEY)
|
| 198 |
+
message = client.messages.create(
|
| 199 |
+
model="claude-sonnet-4-20250514",
|
| 200 |
+
max_tokens=300,
|
| 201 |
+
messages=[{"role": "user", "content": prompt}],
|
| 202 |
+
)
|
| 203 |
+
comment = message.content[0].text.strip() if message.content else ""
|
| 204 |
+
except Exception as e:
|
| 205 |
+
comment = f"[Błąd Claude API: {e}]"
|
| 206 |
+
|
| 207 |
+
return EmotionCommentResponse(comment=comment)
|
| 208 |
+
|
| 209 |
+
|
| 210 |
+
class CompareRequest(BaseModel):
|
| 211 |
+
text: str
|
| 212 |
+
title: str = ""
|
| 213 |
+
source: str = ""
|
| 214 |
+
lang: str = "pl"
|
| 215 |
+
|
| 216 |
+
|
| 217 |
+
class AdapterResult(BaseModel):
|
| 218 |
+
adapter: str
|
| 219 |
+
sentiment: str
|
| 220 |
+
sentiment_score: float
|
| 221 |
+
inference_time_ms: float
|
| 222 |
+
|
| 223 |
+
|
| 224 |
+
class CompareResponse(BaseModel):
|
| 225 |
+
shared_fake_probability: float
|
| 226 |
+
results: List[AdapterResult]
|
| 227 |
+
explanation: str
|
| 228 |
+
|
| 229 |
+
|
| 230 |
+
def _sentiment_only_with_timing(adapter_name: str, text: str, lang: str) -> AdapterResult:
|
| 231 |
+
"""Uruchamia tylko analizę sentymentu i mierzy jej czas."""
|
| 232 |
+
adapter = get_adapter(adapter_name)
|
| 233 |
+
neutral = SENTIMENT_MAP["neutral"].get(lang, "Neutral")
|
| 234 |
+
|
| 235 |
+
if not text or len(text.strip()) < 10:
|
| 236 |
+
print(f"[DEBUG] {adapter_name}: tekst za krótki, zwracam neutral/0.0")
|
| 237 |
+
return AdapterResult(
|
| 238 |
+
adapter=adapter_name,
|
| 239 |
+
sentiment=neutral,
|
| 240 |
+
sentiment_score=0.0,
|
| 241 |
+
inference_time_ms=0.0,
|
| 242 |
+
)
|
| 243 |
+
|
| 244 |
+
start = time.perf_counter()
|
| 245 |
+
try:
|
| 246 |
+
sentiment_result = adapter.analyze_sentiment(text)
|
| 247 |
+
except Exception as exc:
|
| 248 |
+
elapsed = (time.perf_counter() - start) * 1000
|
| 249 |
+
print(f"[DEBUG] {adapter_name}: WYJĄTEK w analyze_sentiment: {type(exc).__name__}: {exc}")
|
| 250 |
+
return AdapterResult(
|
| 251 |
+
adapter=adapter_name,
|
| 252 |
+
sentiment=neutral,
|
| 253 |
+
sentiment_score=0.0,
|
| 254 |
+
inference_time_ms=round(elapsed, 1),
|
| 255 |
+
)
|
| 256 |
+
elapsed_ms = (time.perf_counter() - start) * 1000
|
| 257 |
+
|
| 258 |
+
print(f"[DEBUG] {adapter_name}: sentiment_result = {sentiment_result}")
|
| 259 |
+
|
| 260 |
+
raw_label = sentiment_result.get("label", "neutral")
|
| 261 |
+
raw_score = sentiment_result.get("score", None)
|
| 262 |
+
|
| 263 |
+
print(f"[DEBUG] {adapter_name}: raw_label={raw_label!r} raw_score={raw_score!r} type(score)={type(raw_score)}")
|
| 264 |
+
|
| 265 |
+
sentiment_translated = SENTIMENT_MAP.get(raw_label, {}).get(lang, neutral)
|
| 266 |
+
score_float = round(float(raw_score) if raw_score is not None else 0.0, 4)
|
| 267 |
+
|
| 268 |
+
print(f"[DEBUG] {adapter_name}: → sentiment={sentiment_translated!r} score_float={score_float} time={elapsed_ms:.1f}ms")
|
| 269 |
+
|
| 270 |
+
return AdapterResult(
|
| 271 |
+
adapter=adapter_name,
|
| 272 |
+
sentiment=sentiment_translated,
|
| 273 |
+
sentiment_score=score_float,
|
| 274 |
+
inference_time_ms=round(elapsed_ms, 1),
|
| 275 |
+
)
|
| 276 |
+
|
| 277 |
+
|
| 278 |
+
def _build_prompt(
|
| 279 |
+
request: CompareRequest,
|
| 280 |
+
results: List[AdapterResult],
|
| 281 |
+
shared_fake_probability: float,
|
| 282 |
+
) -> str:
|
| 283 |
+
lang_name = LANG_NAMES.get(request.lang, request.lang)
|
| 284 |
+
text_snippet = request.text[:600] + ("..." if len(request.text) > 600 else "")
|
| 285 |
+
|
| 286 |
+
header_parts = []
|
| 287 |
+
if request.title:
|
| 288 |
+
header_parts.append(f"Tytuł artykułu: {request.title}")
|
| 289 |
+
if request.source:
|
| 290 |
+
header_parts.append(f"Źródło: {request.source}")
|
| 291 |
+
header_parts.append(f"Język tekstu: {lang_name}")
|
| 292 |
+
header_parts.append(f"Prawdopodobieństwo fake news (BART, wspólne dla wszystkich modeli): {shared_fake_probability:.1f}%")
|
| 293 |
+
header_str = "\n".join(header_parts)
|
| 294 |
+
|
| 295 |
+
rows = "\n".join(
|
| 296 |
+
f" {ADAPTER_DESCRIPTIONS[r.adapter]}: {r.sentiment} (pewność: {r.sentiment_score * 100:.0f}%, czas: {r.inference_time_ms:.0f} ms)"
|
| 297 |
+
for r in results
|
| 298 |
+
)
|
| 299 |
+
|
| 300 |
+
sentiments = [r.sentiment for r in results]
|
| 301 |
+
unique_sentiments = set(sentiments)
|
| 302 |
+
agreement = (
|
| 303 |
+
"zgodne" if len(unique_sentiments) == 1
|
| 304 |
+
else f"różne ({', '.join(unique_sentiments)})"
|
| 305 |
+
)
|
| 306 |
+
|
| 307 |
+
lang = request.lang
|
| 308 |
+
response_lang = RESPONSE_LANG.get(lang, "polskim")
|
| 309 |
+
instruction = EXPLAIN_INSTRUCTION.get(lang, EXPLAIN_INSTRUCTION["pl"]).format(
|
| 310 |
+
lang_name=lang_name
|
| 311 |
+
)
|
| 312 |
+
|
| 313 |
+
return f"""Odpowiedz w języku: {response_lang}.
|
| 314 |
+
|
| 315 |
+
{header_str}
|
| 316 |
+
|
| 317 |
+
Fragment tekstu:
|
| 318 |
+
{text_snippet}
|
| 319 |
+
|
| 320 |
+
Wyniki analizy sentymentu:
|
| 321 |
+
{rows}
|
| 322 |
+
|
| 323 |
+
Zgodność modeli: {agreement}
|
| 324 |
+
|
| 325 |
+
{instruction}"""
|
| 326 |
+
|
| 327 |
+
|
| 328 |
+
@router.post("/compare", response_model=CompareResponse)
|
| 329 |
+
def compare_models(request: CompareRequest):
|
| 330 |
+
"""
|
| 331 |
+
Analizuje tekst przez wszystkie 4 adaptery NLP:
|
| 332 |
+
- fake news: jeden wynik ze współdzielonego BART
|
| 333 |
+
- sentyment: każdy adapter osobno z pomiarem czasu
|
| 334 |
+
Zwraca wyniki + wyjaśnienie Claude API.
|
| 335 |
+
"""
|
| 336 |
+
if not request.text or len(request.text.strip()) < 10:
|
| 337 |
+
raise HTTPException(status_code=422, detail="Tekst jest za krótki (min. 10 znaków).")
|
| 338 |
+
|
| 339 |
+
# Fake news — jeden wspólny wynik z BART (uruchamiamy raz)
|
| 340 |
+
try:
|
| 341 |
+
shared_fake_probability = analyze_fake_news_shared(request.text)
|
| 342 |
+
except Exception as e:
|
| 343 |
+
shared_fake_probability = 0.0
|
| 344 |
+
|
| 345 |
+
# Sentyment — każdy adapter osobno z pomiarem czasu
|
| 346 |
+
results: List[AdapterResult] = [
|
| 347 |
+
_sentiment_only_with_timing(name, request.text, request.lang)
|
| 348 |
+
for name in ADAPTER_NAMES
|
| 349 |
+
]
|
| 350 |
+
|
| 351 |
+
# Wyjaśnienie Claude
|
| 352 |
+
explanation = ""
|
| 353 |
+
if ANTHROPIC_API_KEY:
|
| 354 |
+
try:
|
| 355 |
+
client = anthropic.Anthropic(api_key=ANTHROPIC_API_KEY)
|
| 356 |
+
prompt = _build_prompt(request, results, shared_fake_probability)
|
| 357 |
+
message = client.messages.create(
|
| 358 |
+
model="claude-sonnet-4-20250514",
|
| 359 |
+
max_tokens=800,
|
| 360 |
+
messages=[{"role": "user", "content": prompt}],
|
| 361 |
+
)
|
| 362 |
+
explanation = message.content[0].text if message.content else ""
|
| 363 |
+
except Exception as e:
|
| 364 |
+
explanation = f"[Błąd Claude API: {str(e)}]"
|
| 365 |
+
else:
|
| 366 |
+
explanation = "[Brak klucza ANTHROPIC_API_KEY — wyjaśnienie niedostępne]"
|
| 367 |
+
|
| 368 |
+
return CompareResponse(
|
| 369 |
+
shared_fake_probability=shared_fake_probability,
|
| 370 |
+
results=results,
|
| 371 |
+
explanation=explanation,
|
| 372 |
+
)
|
app/routes/misc.py
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Endpointy związane z obsługą dostępnych źródeł wiadomości oraz narzędziami deweloperskimi.
|
| 3 |
+
"""
|
| 4 |
+
|
| 5 |
+
from pathlib import Path
|
| 6 |
+
from typing import Optional
|
| 7 |
+
|
| 8 |
+
from fastapi import APIRouter, Query
|
| 9 |
+
|
| 10 |
+
from ..config import NEWS_FEEDS
|
| 11 |
+
from ..benchmark import run_benchmark, export_json, export_csv, _summary_only
|
| 12 |
+
|
| 13 |
+
router = APIRouter()
|
| 14 |
+
|
| 15 |
+
# Katalog, do którego endpoint zapisuje artefakty benchmarku
|
| 16 |
+
_BENCHMARK_OUT = Path("benchmark_results")
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
@router.get("/sources")
|
| 20 |
+
def get_sources():
|
| 21 |
+
return list(NEWS_FEEDS.keys())
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
@router.get("/benchmark")
|
| 25 |
+
def benchmark(
|
| 26 |
+
adapters: Optional[str] = Query(
|
| 27 |
+
default=None,
|
| 28 |
+
description="Przecinkowa lista adapterów: roberta,xlm-roberta,norbert",
|
| 29 |
+
),
|
| 30 |
+
langs: Optional[str] = Query(
|
| 31 |
+
default=None,
|
| 32 |
+
description="Przecinkowa lista języków: en,pl,no",
|
| 33 |
+
),
|
| 34 |
+
save: bool = Query(
|
| 35 |
+
default=True,
|
| 36 |
+
description="Czy zapisać wyniki do JSON i CSV w katalogu benchmark_results/",
|
| 37 |
+
),
|
| 38 |
+
full: bool = Query(
|
| 39 |
+
default=False,
|
| 40 |
+
description="Czy zwrócić szczegółowe wyniki per_text (domyślnie tylko podsumowanie)",
|
| 41 |
+
),
|
| 42 |
+
):
|
| 43 |
+
"""
|
| 44 |
+
Uruchamia benchmark NLP i zwraca wyniki.
|
| 45 |
+
|
| 46 |
+
Czas odpowiedzi zależy od liczby adapterów i języków — może wynosić kilkadziesiąt sekund
|
| 47 |
+
przy pierwszym uruchomieniu (lazy-loading modeli HuggingFace).
|
| 48 |
+
|
| 49 |
+
Przykłady:
|
| 50 |
+
- GET /benchmark
|
| 51 |
+
- GET /benchmark?adapters=roberta,xlm-roberta&langs=en,pl
|
| 52 |
+
- GET /benchmark?full=true&save=false
|
| 53 |
+
"""
|
| 54 |
+
adapter_names = [a.strip() for a in adapters.split(",")] if adapters else None
|
| 55 |
+
lang_list = [l.strip() for l in langs.split(",")] if langs else None
|
| 56 |
+
|
| 57 |
+
results = run_benchmark(adapter_names=adapter_names, langs=lang_list)
|
| 58 |
+
|
| 59 |
+
if save:
|
| 60 |
+
export_json(results, _BENCHMARK_OUT / "benchmark.json")
|
| 61 |
+
export_csv(results, _BENCHMARK_OUT / "benchmark.csv")
|
| 62 |
+
|
| 63 |
+
return {
|
| 64 |
+
"results": results if full else _summary_only(results),
|
| 65 |
+
"saved": save,
|
| 66 |
+
"output_dir": str(_BENCHMARK_OUT) if save else None,
|
| 67 |
+
}
|
| 68 |
+
|
app/routes/news.py
ADDED
|
@@ -0,0 +1,266 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Endpointy związane z pobieraniem, analizą i strumieniowaniem wiadomości.
|
| 3 |
+
"""
|
| 4 |
+
|
| 5 |
+
import asyncio
|
| 6 |
+
import json
|
| 7 |
+
from typing import List
|
| 8 |
+
|
| 9 |
+
from fastapi import APIRouter, HTTPException
|
| 10 |
+
from fastapi.responses import StreamingResponse
|
| 11 |
+
from fastapi_cache.decorator import cache
|
| 12 |
+
|
| 13 |
+
from ..config import CACHE_TTL, NEWS_FEEDS
|
| 14 |
+
from ..rss_utils import fetch_feed, clean_html, get_from_cache, set_to_cache, FeedFetchError
|
| 15 |
+
from ..nlp_service import analyze_news, set_active_adapter
|
| 16 |
+
|
| 17 |
+
router = APIRouter()
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
@router.get("/news/{source}")
|
| 21 |
+
def get_news(source: str, lang: str = "pl", model: str = "roberta"):
|
| 22 |
+
# Walidacja źródła
|
| 23 |
+
if source not in NEWS_FEEDS:
|
| 24 |
+
raise HTTPException(status_code=404, detail="Źródło nieobsługiwane")
|
| 25 |
+
|
| 26 |
+
# Ustawienie aktywnego adaptera NLP
|
| 27 |
+
try:
|
| 28 |
+
set_active_adapter(model)
|
| 29 |
+
except ValueError:
|
| 30 |
+
raise HTTPException(status_code=400, detail=f"Nieznany model: {model}")
|
| 31 |
+
|
| 32 |
+
url = NEWS_FEEDS[source]
|
| 33 |
+
|
| 34 |
+
# Pobranie RSS — FeedFetchError (404/timeout) zwraca pustą listę zamiast HTTP 500
|
| 35 |
+
try:
|
| 36 |
+
feed = fetch_feed(url)
|
| 37 |
+
except FeedFetchError as e:
|
| 38 |
+
print(f"⚠️ Feed niedostępny [{source}]: {e}")
|
| 39 |
+
return {"source": source, "articles": [], "error": str(e)}
|
| 40 |
+
except Exception as e:
|
| 41 |
+
raise HTTPException(status_code=500, detail=f"Błąd pobierania newsów: {str(e)}")
|
| 42 |
+
|
| 43 |
+
if not feed.entries:
|
| 44 |
+
return {"source": source, "articles": []}
|
| 45 |
+
|
| 46 |
+
MAX_ARTICLES = 5
|
| 47 |
+
articles: List[dict] = []
|
| 48 |
+
|
| 49 |
+
# Przetwarzanie i analiza artykułów
|
| 50 |
+
for entry in feed.entries[:MAX_ARTICLES]:
|
| 51 |
+
summary = clean_html(entry.get("summary", "Brak opisu"))
|
| 52 |
+
text_to_analyze = (summary or "").strip() or entry.get("title", "")
|
| 53 |
+
analysis = analyze_news(text_to_analyze, lang)
|
| 54 |
+
|
| 55 |
+
articles.append({
|
| 56 |
+
"title": entry.get("title", "Bez tytułu"),
|
| 57 |
+
"link": entry.get("link", ""),
|
| 58 |
+
"summary": summary,
|
| 59 |
+
"published": entry.get("published", "Brak daty"),
|
| 60 |
+
"source": source,
|
| 61 |
+
"model": model,
|
| 62 |
+
**analysis
|
| 63 |
+
})
|
| 64 |
+
|
| 65 |
+
return {"source": source, "articles": articles}
|
| 66 |
+
|
| 67 |
+
|
| 68 |
+
# Strumieniowanie newsów przez SSE
|
| 69 |
+
@router.get("/stream-news/{source}")
|
| 70 |
+
async def stream_news(source: str, lang: str = "pl", model: str = "roberta"):
|
| 71 |
+
if source not in NEWS_FEEDS:
|
| 72 |
+
raise HTTPException(status_code=404, detail="Źródło nieobsługiwane")
|
| 73 |
+
|
| 74 |
+
# Walidacja i ustawienie adaptera przed uruchomieniem generatora
|
| 75 |
+
try:
|
| 76 |
+
set_active_adapter(model)
|
| 77 |
+
except ValueError:
|
| 78 |
+
raise HTTPException(status_code=400, detail=f"Nieznany model: {model}")
|
| 79 |
+
|
| 80 |
+
async def event_generator():
|
| 81 |
+
# Próba pobrania RSS z cache
|
| 82 |
+
try:
|
| 83 |
+
cached = get_from_cache(source)
|
| 84 |
+
if cached:
|
| 85 |
+
feed = cached
|
| 86 |
+
else:
|
| 87 |
+
feed = fetch_feed(NEWS_FEEDS[source])
|
| 88 |
+
set_to_cache(source, feed)
|
| 89 |
+
except FeedFetchError as e:
|
| 90 |
+
print(f"⚠️ Feed niedostępny [{source}]: {e}")
|
| 91 |
+
yield f"event: backend_error\ndata: {json.dumps({'message': str(e), 'status': e.status_code})}\n\n"
|
| 92 |
+
yield f"event: done\ndata: {json.dumps({'count': 0})}\n\n"
|
| 93 |
+
return
|
| 94 |
+
except Exception as e:
|
| 95 |
+
yield f"event: backend_error\ndata: {json.dumps({'message': str(e)})}\n\n"
|
| 96 |
+
yield f"event: done\ndata: {json.dumps({'count': 0})}\n\n"
|
| 97 |
+
return
|
| 98 |
+
|
| 99 |
+
entries = feed.entries or []
|
| 100 |
+
MAX_ARTICLES = 5
|
| 101 |
+
to_send = entries[:MAX_ARTICLES]
|
| 102 |
+
|
| 103 |
+
# Metadane dla klienta
|
| 104 |
+
yield f"event: meta\ndata: {json.dumps({'total': len(to_send)})}\n\n"
|
| 105 |
+
|
| 106 |
+
sent = 0
|
| 107 |
+
for entry in to_send:
|
| 108 |
+
summary = clean_html(entry.get("summary", "Brak opisu"))
|
| 109 |
+
text_to_analyze = (summary or "").strip() or entry.get("title", "")
|
| 110 |
+
|
| 111 |
+
# Analiza NLP uruchamiana w executorze (CPU-bound)
|
| 112 |
+
loop = asyncio.get_event_loop()
|
| 113 |
+
analysis = await loop.run_in_executor(
|
| 114 |
+
None, lambda: analyze_news(text_to_analyze, lang)
|
| 115 |
+
)
|
| 116 |
+
|
| 117 |
+
article = {
|
| 118 |
+
"title": entry.get("title", "Bez tytułu"),
|
| 119 |
+
"link": entry.get("link", ""),
|
| 120 |
+
"summary": summary,
|
| 121 |
+
"published": entry.get("published", "Brak daty"),
|
| 122 |
+
"source": source,
|
| 123 |
+
"model": model,
|
| 124 |
+
**analysis,
|
| 125 |
+
}
|
| 126 |
+
|
| 127 |
+
yield f"data: {json.dumps(article, ensure_ascii=False)}\n\n"
|
| 128 |
+
sent += 1
|
| 129 |
+
await asyncio.sleep(0.3)
|
| 130 |
+
|
| 131 |
+
yield f"event: done\ndata: {json.dumps({'count': sent})}\n\n"
|
| 132 |
+
|
| 133 |
+
return StreamingResponse(
|
| 134 |
+
event_generator(),
|
| 135 |
+
media_type="text/event-stream",
|
| 136 |
+
headers={
|
| 137 |
+
"Content-Type": "text/event-stream",
|
| 138 |
+
"Access-Control-Allow-Origin": "*",
|
| 139 |
+
"Access-Control-Allow-Headers": "*",
|
| 140 |
+
"Cache-Control": "no-cache",
|
| 141 |
+
"X-Accel-Buffering": "no",
|
| 142 |
+
},
|
| 143 |
+
)
|
| 144 |
+
|
| 145 |
+
|
| 146 |
+
@router.get("/api/charts/summary")
|
| 147 |
+
@cache(expire=CACHE_TTL)
|
| 148 |
+
async def get_charts_summary(lang: str = "pl"):
|
| 149 |
+
"""
|
| 150 |
+
Szybkie statystyki zbiorcze dla wszystkich źródeł
|
| 151 |
+
(uproszczona analiza oparta na tytułach).
|
| 152 |
+
"""
|
| 153 |
+
sources = [
|
| 154 |
+
"BBC", "CNN", "NYTimes", "Guardian", "AlJazeera",
|
| 155 |
+
"PolsatNews", "Money", "Bankier", "SpidersWeb", "GazetaPrawna"
|
| 156 |
+
]
|
| 157 |
+
|
| 158 |
+
summary_data = {}
|
| 159 |
+
|
| 160 |
+
for source in sources:
|
| 161 |
+
if source not in NEWS_FEEDS:
|
| 162 |
+
continue
|
| 163 |
+
|
| 164 |
+
try:
|
| 165 |
+
feed = fetch_feed(NEWS_FEEDS[source])
|
| 166 |
+
|
| 167 |
+
if not feed.entries:
|
| 168 |
+
summary_data[source] = {
|
| 169 |
+
"count": 0,
|
| 170 |
+
"emotions": {"Pozytywne": 0, "Negatywne": 0, "Neutralne": 0}
|
| 171 |
+
}
|
| 172 |
+
continue
|
| 173 |
+
|
| 174 |
+
# Analiza tylko kilku tytułów (szybko)
|
| 175 |
+
articles = feed.entries[:3]
|
| 176 |
+
emotion_counts = {"Pozytywne": 0, "Negatywne": 0, "Neutralne": 0}
|
| 177 |
+
|
| 178 |
+
for entry in articles:
|
| 179 |
+
title = entry.get("title", "").lower()
|
| 180 |
+
if any(word in title for word in ["good", "positive", "gain", "up", "success", "dobry", "wzrost", "zysk"]):
|
| 181 |
+
emotion_counts["Pozytywne"] += 1
|
| 182 |
+
elif any(word in title for word in ["bad", "negative", "fall", "down", "loss", "crisis", "zły", "spadek", "kryzys"]):
|
| 183 |
+
emotion_counts["Negatywne"] += 1
|
| 184 |
+
else:
|
| 185 |
+
emotion_counts["Neutralne"] += 1
|
| 186 |
+
|
| 187 |
+
summary_data[source] = {
|
| 188 |
+
"count": len(feed.entries),
|
| 189 |
+
"analyzed": len(articles),
|
| 190 |
+
"emotions": emotion_counts,
|
| 191 |
+
"latest_title": articles[0].get("title", "")[:50] if articles else ""
|
| 192 |
+
}
|
| 193 |
+
|
| 194 |
+
except Exception:
|
| 195 |
+
summary_data[source] = {
|
| 196 |
+
"count": 0,
|
| 197 |
+
"emotions": {"Pozytywne": 0, "Negatywne": 0, "Neutralne": 0}
|
| 198 |
+
}
|
| 199 |
+
|
| 200 |
+
return {
|
| 201 |
+
"summary": summary_data,
|
| 202 |
+
"total_sources": len(summary_data),
|
| 203 |
+
"cache_ttl": CACHE_TTL
|
| 204 |
+
}
|
| 205 |
+
|
| 206 |
+
|
| 207 |
+
@router.get("/emotion-stats/{source}")
|
| 208 |
+
@cache(expire=CACHE_TTL)
|
| 209 |
+
def get_emotion_stats(source: str, lang: str = "pl", model: str = "roberta"):
|
| 210 |
+
# Statystyki emocji dla jednego źródła
|
| 211 |
+
if source not in NEWS_FEEDS:
|
| 212 |
+
raise HTTPException(status_code=404, detail="Źródło nieobsługiwane")
|
| 213 |
+
|
| 214 |
+
try:
|
| 215 |
+
news_data = get_news(source, lang, model)
|
| 216 |
+
articles = news_data.get("articles", [])
|
| 217 |
+
except Exception as e:
|
| 218 |
+
raise HTTPException(status_code=500, detail=f"Błąd generowania statystyk: {str(e)}")
|
| 219 |
+
|
| 220 |
+
emotion_counts = {"Pozytywne": 0, "Negatywne": 0, "Neutralne": 0}
|
| 221 |
+
|
| 222 |
+
for article in articles:
|
| 223 |
+
sentiment = article.get("sentiment", "Neutralne")
|
| 224 |
+
if sentiment in emotion_counts:
|
| 225 |
+
emotion_counts[sentiment] += 1
|
| 226 |
+
|
| 227 |
+
total = len(articles)
|
| 228 |
+
emotion_percentages = {
|
| 229 |
+
emotion: (round((count / total) * 100, 2) if total > 0 else 0)
|
| 230 |
+
for emotion, count in emotion_counts.items()
|
| 231 |
+
}
|
| 232 |
+
|
| 233 |
+
return {
|
| 234 |
+
"source": source,
|
| 235 |
+
"total_articles": total,
|
| 236 |
+
"emotion_counts": emotion_counts,
|
| 237 |
+
"emotion_percentages": emotion_percentages
|
| 238 |
+
}
|
| 239 |
+
|
| 240 |
+
|
| 241 |
+
@router.get("/charts-data")
|
| 242 |
+
@cache(expire=CACHE_TTL)
|
| 243 |
+
async def get_all_charts_data(lang: str = "pl"):
|
| 244 |
+
"""
|
| 245 |
+
Kompatybilność ze starszym frontendem – agreguje dane ze wszystkich źródeł.
|
| 246 |
+
"""
|
| 247 |
+
sources = [
|
| 248 |
+
"BBC", "CNN", "NYTimes", "Guardian", "AlJazeera",
|
| 249 |
+
"PolsatNews", "Money", "Bankier", "SpidersWeb", "GazetaPrawna"
|
| 250 |
+
]
|
| 251 |
+
|
| 252 |
+
async def fetch_stats(source):
|
| 253 |
+
try:
|
| 254 |
+
return {source: await get_emotion_stats(source, lang)}
|
| 255 |
+
except Exception:
|
| 256 |
+
return {source: None}
|
| 257 |
+
|
| 258 |
+
tasks = [fetch_stats(source) for source in sources]
|
| 259 |
+
results = await asyncio.gather(*tasks, return_exceptions=True)
|
| 260 |
+
|
| 261 |
+
all_data = {}
|
| 262 |
+
for result in results:
|
| 263 |
+
if isinstance(result, dict):
|
| 264 |
+
all_data.update(result)
|
| 265 |
+
|
| 266 |
+
return {"charts": all_data}
|
app/routes/saved.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Endpointy związane z zapisywaniem i zarządzaniem zapisanymi artykułami.
|
| 3 |
+
"""
|
| 4 |
+
|
| 5 |
+
from fastapi import APIRouter, HTTPException, Request
|
| 6 |
+
|
| 7 |
+
from ..storage import read_all, append_article, delete_by_title
|
| 8 |
+
from ..models import Article
|
| 9 |
+
|
| 10 |
+
router = APIRouter()
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
@router.get("/saved-articles")
|
| 14 |
+
def get_saved_articles():
|
| 15 |
+
# Zwraca listę wszystkich zapisanych artykułów
|
| 16 |
+
return read_all()
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
@router.post("/save-article")
|
| 20 |
+
def save_article(article: Article):
|
| 21 |
+
# Zapisuje nowy artykuł do magazynu danych
|
| 22 |
+
return append_article(article.dict())
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
@router.delete("/delete-article")
|
| 26 |
+
async def delete_article(request: Request):
|
| 27 |
+
# Usuwa artykuł na podstawie tytułu przekazanego w body requestu
|
| 28 |
+
data = await request.json()
|
| 29 |
+
title = data.get("title")
|
| 30 |
+
|
| 31 |
+
if not title:
|
| 32 |
+
raise HTTPException(status_code=400, detail="Brak pola 'title'")
|
| 33 |
+
|
| 34 |
+
return delete_by_title(title)
|
app/rss_utils.py
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Narzędzia pomocnicze do pobierania i przetwarzania kanałów RSS oraz cache w pamięci.
|
| 3 |
+
"""
|
| 4 |
+
|
| 5 |
+
import time
|
| 6 |
+
import requests
|
| 7 |
+
import feedparser
|
| 8 |
+
from bs4 import BeautifulSoup
|
| 9 |
+
from typing import Dict, Any
|
| 10 |
+
|
| 11 |
+
from .config import CACHE_TTL_SECONDS
|
| 12 |
+
|
| 13 |
+
# Prosty cache w pamięci (key -> (value, timestamp))
|
| 14 |
+
_cache_data: Dict[str, tuple[Any, float]] = {}
|
| 15 |
+
|
| 16 |
+
|
| 17 |
+
def clean_html(text: str) -> str:
|
| 18 |
+
# Usuwa znaczniki HTML z treści RSS
|
| 19 |
+
return BeautifulSoup(text or "", "html.parser").get_text()
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
def get_from_cache(key: str):
|
| 23 |
+
# Pobiera dane z cache, jeśli nie przekroczyły TTL
|
| 24 |
+
entry = _cache_data.get(key)
|
| 25 |
+
if not entry:
|
| 26 |
+
return None
|
| 27 |
+
|
| 28 |
+
value, ts = entry
|
| 29 |
+
if time.time() - ts > CACHE_TTL_SECONDS:
|
| 30 |
+
del _cache_data[key]
|
| 31 |
+
return None
|
| 32 |
+
|
| 33 |
+
return value
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
def set_to_cache(key: str, value):
|
| 37 |
+
# Zapisuje dane do cache wraz z timestampem
|
| 38 |
+
_cache_data[key] = (value, time.time())
|
| 39 |
+
|
| 40 |
+
|
| 41 |
+
class FeedFetchError(Exception):
|
| 42 |
+
"""Wyjątek rzucany gdy feed RSS jest niedostępny lub zwraca błąd HTTP."""
|
| 43 |
+
def __init__(self, url: str, status_code: int, message: str = ""):
|
| 44 |
+
self.url = url
|
| 45 |
+
self.status_code = status_code
|
| 46 |
+
super().__init__(message or f"Feed unavailable: HTTP {status_code} for {url}")
|
| 47 |
+
|
| 48 |
+
|
| 49 |
+
def fetch_feed(url: str):
|
| 50 |
+
"""
|
| 51 |
+
Pobiera i parsuje kanał RSS z ustawionym User-Agent.
|
| 52 |
+
Rzuca FeedFetchError jeśli serwer zwróci błąd HTTP (4xx/5xx),
|
| 53 |
+
co pozwala wywołującemu zwrócić pustą listę zamiast HTTP 500.
|
| 54 |
+
"""
|
| 55 |
+
headers = {
|
| 56 |
+
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
|
| 57 |
+
"Accept": "application/rss+xml, application/xml, text/xml, */*",
|
| 58 |
+
}
|
| 59 |
+
try:
|
| 60 |
+
resp = requests.get(url, timeout=7, headers=headers)
|
| 61 |
+
except requests.exceptions.Timeout:
|
| 62 |
+
raise FeedFetchError(url, 408, f"Timeout fetching feed: {url}")
|
| 63 |
+
except requests.exceptions.ConnectionError as e:
|
| 64 |
+
raise FeedFetchError(url, 503, f"Connection error for {url}: {e}")
|
| 65 |
+
|
| 66 |
+
if resp.status_code >= 400:
|
| 67 |
+
raise FeedFetchError(url, resp.status_code,
|
| 68 |
+
f"HTTP {resp.status_code} for feed: {url}")
|
| 69 |
+
|
| 70 |
+
return feedparser.parse(resp.content)
|
app/storage.py
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Warstwa dostępu do danych dla zapisanych artykułów (plik JSON).
|
| 3 |
+
"""
|
| 4 |
+
|
| 5 |
+
import json
|
| 6 |
+
import threading
|
| 7 |
+
from fastapi import HTTPException
|
| 8 |
+
|
| 9 |
+
from .config import SAVED_FILE
|
| 10 |
+
|
| 11 |
+
# Blokada wątków dla operacji zapisu/odczytu
|
| 12 |
+
_lock = threading.Lock()
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
def ensure_file():
|
| 16 |
+
# Tworzy plik danych, jeśli nie istnieje
|
| 17 |
+
if not SAVED_FILE.exists():
|
| 18 |
+
SAVED_FILE.write_text("[]", encoding="utf-8")
|
| 19 |
+
|
| 20 |
+
|
| 21 |
+
def read_all():
|
| 22 |
+
# Zwraca wszystkie zapisane artykuły
|
| 23 |
+
ensure_file()
|
| 24 |
+
try:
|
| 25 |
+
return json.loads(SAVED_FILE.read_text(encoding="utf-8"))
|
| 26 |
+
except json.JSONDecodeError:
|
| 27 |
+
return []
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
def append_article(article: dict):
|
| 31 |
+
# Dodaje nowy artykuł do pliku JSON
|
| 32 |
+
ensure_file()
|
| 33 |
+
try:
|
| 34 |
+
with _lock, open(SAVED_FILE, "r+", encoding="utf-8") as f:
|
| 35 |
+
data = json.load(f)
|
| 36 |
+
data.append(article)
|
| 37 |
+
f.seek(0)
|
| 38 |
+
json.dump(data, f, ensure_ascii=False, indent=4)
|
| 39 |
+
|
| 40 |
+
return {"message": "Artykuł zapisany pomyślnie."}
|
| 41 |
+
except Exception as e:
|
| 42 |
+
raise HTTPException(status_code=500, detail=str(e))
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
def delete_by_title(title: str):
|
| 46 |
+
# Usuwa artykuł na podstawie tytułu
|
| 47 |
+
ensure_file()
|
| 48 |
+
try:
|
| 49 |
+
with _lock, open(SAVED_FILE, "r+", encoding="utf-8") as f:
|
| 50 |
+
saved = json.load(f)
|
| 51 |
+
new_saved = [
|
| 52 |
+
a for a in saved if a.get("title") != title
|
| 53 |
+
]
|
| 54 |
+
f.seek(0)
|
| 55 |
+
f.truncate()
|
| 56 |
+
json.dump(new_saved, f, ensure_ascii=False, indent=4)
|
| 57 |
+
|
| 58 |
+
return {"message": "Artykuł usunięty."}
|
| 59 |
+
except Exception as e:
|
| 60 |
+
raise HTTPException(status_code=500, detail=str(e))
|
requirements.txt
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Zależności backendu aplikacji TruthScan
|
| 2 |
+
|
| 3 |
+
# Framework API i serwer ASGI
|
| 4 |
+
fastapi==0.115.12
|
| 5 |
+
uvicorn==0.34.3
|
| 6 |
+
starlette==0.46.2
|
| 7 |
+
|
| 8 |
+
# Walidacja danych i modele
|
| 9 |
+
pydantic==2.11.7
|
| 10 |
+
pydantic_core==2.33.2
|
| 11 |
+
annotated-types==0.7.0
|
| 12 |
+
typing-extensions>=4.12.2
|
| 13 |
+
|
| 14 |
+
# Cache
|
| 15 |
+
fastapi-cache2==0.2.2
|
| 16 |
+
|
| 17 |
+
# NLP i uczenie maszynowe
|
| 18 |
+
transformers==4.44.2
|
| 19 |
+
tokenizers==0.19.1
|
| 20 |
+
torch==2.3.1+cpu --find-links https://download.pytorch.org/whl/cpu
|
| 21 |
+
|
| 22 |
+
# Przetwarzanie RSS i HTML
|
| 23 |
+
feedparser==6.0.11
|
| 24 |
+
beautifulsoup4==4.12.3
|
| 25 |
+
requests==2.31.0
|
| 26 |
+
|
| 27 |
+
# Konfiguracja środowiska
|
| 28 |
+
python-dotenv==1.0.1
|
| 29 |
+
|
| 30 |
+
# Claude API (Anthropic)
|
| 31 |
+
anthropic==0.40.0
|