File size: 2,444 Bytes
adf912b
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
"""内容审核台:给一批评论打 有害/垃圾/需人工复核 的校准概率,并按风险排序。
用法: python apps/moderator.py [file.txt]   # 每行一条评论;无参数用内置样例
"""
import sys, time
from common import get_agent, bar

QUESTIONS = {
    "toxic": {"type": "noul", "instructions": "Is this comment abusive, hateful or harassing toward someone?"},
    "spam": {"type": "noul", "instructions": "Is this comment spam or unsolicited advertising?"},
    "topic": {"type": "choice", "instructions": "What is the comment mainly about?",
              "criteria": {"product": "the product or service itself", "politics": "political opinion",
                           "personal": "attacks or remarks about a person", "offtopic": "unrelated chatter"}},
    "severity": {"type": "score", "instructions": "How severe is the policy violation, if any?",
                 "criteria": ["none", "mild", "serious", "ban-worthy"]},
}

SAMPLES = [
    "This update is great, the new editor is so much faster!",
    "Buy cheap followers now!!! visit my profile link, 50% off today only",
    "You are a worthless idiot and everyone here knows it.",
    "这个功能真的太难用了,建议回滚到上个版本。",
    "滚出去,你这种垃圾不配在这发言。",
    "Honestly both parties are the same, nothing will change.",
    "Does anyone know if the API supports webhooks?",
]

def main():
    agent = get_agent("multilingual")
    texts = [l.strip() for l in open(sys.argv[1], encoding="utf-8") if l.strip()] if len(sys.argv) > 1 else SAMPLES
    t = time.time()
    results = [agent.predict({"comment": c}, QUESTIONS) for c in texts]
    dt = time.time() - t
    rows = []
    for c, r in zip(texts, results):
        a = r["answers"]
        risk = max(a["toxic"]["noul"], a["spam"]["noul"])
        rows.append((risk, a["toxic"]["noul"], a["spam"]["noul"],
                     a["topic"]["choice"], round(a["severity"]["score"],1), c))
    rows.sort(reverse=True)
    print(f"{len(texts)} comments in {dt*1000:.0f} ms  ({dt*1000/len(texts):.0f} ms each)\n")
    print(f"{'risk':>5} {'toxic':>5} {'spam':>5}  {'topic':9} {'sev':>4}  comment")
    for risk, tox, spam, topic, sev, c in rows:
        flag = "🚨" if risk > 0.7 else ("⚠️ " if risk > 0.4 else "  ")
        print(f"{flag}{risk:5.2f} {tox:5.2f} {spam:5.2f}  {topic:9} {str(sev):>4}  {c[:60]}")

if __name__ == "__main__":
    main()