Evkoff commited on
Commit
bd84d11
Β·
verified Β·
1 Parent(s): 7e8fd8e

Update app.py

Browse files
Files changed (1) hide show
  1. app.py +53 -8
app.py CHANGED
@@ -1,12 +1,22 @@
1
- """Minimal Gradio app β€” a deployment probe, not the product yet.
2
 
3
- No models are loaded, so a failure here points at the Space rather than at
4
- anything this project computes.
 
 
 
 
 
 
5
  """
6
 
7
  import gradio as gr
8
  import spaces
9
 
 
 
 
 
10
 
11
  # A ZeroGPU Space refuses to start without at least one @spaces.GPU function:
12
  # the first attempt failed with "No @spaces.GPU function detected during
@@ -16,17 +26,45 @@ import spaces
16
  # So the decorator sits on a function that satisfies the requirement without
17
  # taking part in the work. Decorating the real check instead would spend the
18
  # 5-minute daily GPU quota and add a queue wait before every request, in
19
- # exchange for no speed-up at all β€” and moving models to CUDA at module level,
20
- # as ZeroGPU prefers, would break running this file on a laptop with no CUDA.
21
  @spaces.GPU(duration=10)
22
  def _zerogpu_requirement() -> None:
23
  """Not called. Present so the platform allows the Space to start."""
24
  return None
25
 
26
 
 
 
 
 
 
 
 
 
 
27
  def check(source: str, response: str) -> str:
28
- """Placeholder for the real detectors β€” reports sizes and nothing more."""
29
- return f"source: {len(source)} characters\nresponse: {len(response)} characters"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
30
 
31
 
32
  demo = gr.Interface(
@@ -37,7 +75,14 @@ demo = gr.Interface(
37
  ],
38
  outputs=gr.Textbox(label="Result"),
39
  title="AI Output Auditor",
40
- description="Deployment probe. Detectors are not wired in yet.",
 
 
 
 
 
 
 
41
  )
42
 
43
  # SSR is on by default and shuts the app down immediately on Spaces β€” a known
 
1
+ """AI Output Auditor β€” the two local detectors, live on the Space.
2
 
3
+ Not the finished product yet: the LLM judge needs an API key in the Space's
4
+ secrets and is wired in separately.
5
+
6
+ What this file settles is the one thing that has been in doubt since the Space
7
+ was created β€” whether HHEM loads and scores correctly inside the free ZeroGPU
8
+ image. Loading is not enough: a model can load with freshly initialised
9
+ weights and return confident nonsense, which is why the acceptance test is a
10
+ number and not a green light.
11
  """
12
 
13
  import gradio as gr
14
  import spaces
15
 
16
+ from src.embeddings import EmbeddingDetector
17
+ from src.entailment import EntailmentDetector
18
+ from src.halueval import TestCase
19
+
20
 
21
  # A ZeroGPU Space refuses to start without at least one @spaces.GPU function:
22
  # the first attempt failed with "No @spaces.GPU function detected during
 
26
  # So the decorator sits on a function that satisfies the requirement without
27
  # taking part in the work. Decorating the real check instead would spend the
28
  # 5-minute daily GPU quota and add a queue wait before every request, in
29
+ # exchange for no speed-up at all.
 
30
  @spaces.GPU(duration=10)
31
  def _zerogpu_requirement() -> None:
32
  """Not called. Present so the platform allows the Space to start."""
33
  return None
34
 
35
 
36
+ # Built once at import time rather than inside check(). About 530 MB of weights
37
+ # download on a cold start, and doing it here puts that wait on the Space's own
38
+ # startup instead of on whoever opens the page first. It also means a broken
39
+ # model shows up as a failed startup, which is far easier to diagnose than a
40
+ # request that mysteriously errors.
41
+ embeddings = EmbeddingDetector()
42
+ entailment = EntailmentDetector()
43
+
44
+
45
  def check(source: str, response: str) -> str:
46
+ """Run both local detectors on one source/response pair."""
47
+ # TestCase was designed for evaluation, where the correct answer is known.
48
+ # Here it is not β€” that is the entire question the user is asking β€” so the
49
+ # label is empty and the id is a placeholder. Worth revisiting later: a
50
+ # product should not have to invent evaluation metadata to ask a detector
51
+ # a question.
52
+ case = TestCase(
53
+ case_id="live",
54
+ subset="live",
55
+ source=source,
56
+ response=response,
57
+ label="",
58
+ )
59
+
60
+ lines = []
61
+ for name, detector in (("Embeddings", embeddings), ("HHEM", entailment)):
62
+ result = detector.check(case)
63
+ lines.append(
64
+ f"{name}: {result.verdict} "
65
+ f"(score {result.score:.4f}, {result.latency_ms:.0f} ms)"
66
+ )
67
+ return "\n".join(lines)
68
 
69
 
70
  demo = gr.Interface(
 
75
  ],
76
  outputs=gr.Textbox(label="Result"),
77
  title="AI Output Auditor",
78
+ description="Two local detectors. The LLM judge is not wired in yet.",
79
+ # The reference pair, so the acceptance test is one click. Measured on a
80
+ # laptop with these exact sentences: the Berlin row gives 0.7573 and
81
+ # 0.0078, the Paris row 1.0000 and 0.8467. The Space must reproduce them.
82
+ examples=[
83
+ ["The capital of France is Paris.", "The capital of France is Berlin."],
84
+ ["The capital of France is Paris.", "The capital of France is Paris."],
85
+ ],
86
  )
87
 
88
  # SSR is on by default and shuts the app down immediately on Spaces β€” a known