File size: 2,379 Bytes
adf912b
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
"""Hacker News 雷达:拉取实时热帖标题,用 laya 给每条打 主题/是否硬核技术/是否值得读 标签。
用法: python apps/hn_radar.py [N=30]
"""
import os, sys, json, time, urllib.request
from common import get_agent, bar

QUESTIONS = {
    "topic": {"type": "choice", "instructions": "What is this Hacker News story about?",
              "criteria": {"ai": "machine learning, LLMs, models", "systems": "OS, compilers, databases, hardware, GPUs",
                           "web": "frontend, browsers, web frameworks", "security": "vulnerabilities, hacking, privacy",
                           "business": "startups, funding, layoffs, policy", "science": "physics, biology, space, math",
                           "other": "anything else"}},
    "technical_depth": {"type": "score", "instructions": "How technically deep is this likely to be?",
                        "criteria": ["fluff", "medium", "deep dive"]},
    "showhn": {"type": "noul", "instructions": "Is this a project someone built and is showing off?"},
}

def fetch(n):
    """One request to the HN Algolia API (front page)."""
    r = json.load(urllib.request.urlopen(f"https://hn.algolia.com/api/v1/search?tags=front_page&hitsPerPage={n}", timeout=20))
    return [{"title": h["title"], "url": h.get("url") or "", "score": h.get("points", 0)} for h in r["hits"] if h.get("title")]

def main():
    n = int(sys.argv[1]) if len(sys.argv) > 1 else 30
    print(f"fetching {n} HN top stories...")
    items = fetch(n)
    agent = get_agent(os.environ.get("LAYA_VARIANT", "multilingual"))
    t = time.time()
    res = [agent.predict({"title": it["title"], "url": it.get("url", "")}, QUESTIONS) for it in items]
    dt = time.time() - t
    print(f"classified {len(items)} in {dt*1000:.0f} ms\n")
    by_topic = {}
    for it, r in zip(items, res):
        a = r["answers"]
        by_topic.setdefault(a["topic"]["choice"], []).append((round(a["technical_depth"]["score"],1), a["showhn"]["noul"], it))
    for topic, lst in sorted(by_topic.items(), key=lambda kv: -len(kv[1])):
        print(f"## {topic}  ({len(lst)})")
        for depth, show_p, it in sorted(lst, key=lambda x: -(x[0] or 0)):
            tag = " [show]" if show_p > 0.5 else ""
            print(f"  depth={depth} ↑{it.get('score',0):<4} {it['title'][:70]}{tag}")
        print()

if __name__ == "__main__":
    main()