betterwithage commited on
Commit
52875c6
·
verified ·
1 Parent(s): a46ed13

chore(sync): mirror backend .py + Dockerfile to Space (hf-sync-backend)

Browse files

Automated backend sync from szl-holdings/a11oy main via hf-sync-backend.
Updated (differed from the Space): Dockerfile, szl_trajectory_sign.py
Deleted (gone from the repo + Dockerfile COPY set): (none)

Keeps the Space-built backend (serve.py + the Dockerfile-COPY'd .py
modules) identical to GitHub main so the Space never rebuilds from a
stale backend, new endpoints don't 404 there, and orphaned modules
removed from the repo don't linger in the Space tree.

Files changed (2) hide show
  1. Dockerfile +5 -0
  2. szl_trajectory_sign.py +310 -0
Dockerfile CHANGED
@@ -153,6 +153,11 @@ COPY szl_observability.py ./
153
  # SZLHOLDINGS/a11oy-verifiable-corpus. Per-file COPY (this Dockerfile never uses
154
  # `COPY . .`); without it the lazy import is a no-op and receipts never publish.
155
  COPY szl_corpus_publish.py ./
 
 
 
 
 
156
 
157
  # Copy serve orchestrator and gates manifest
158
  # ADDITIVE (live-ops): orchestration + AI-observability module — per-file COPY Dockerfile
 
153
  # SZLHOLDINGS/a11oy-verifiable-corpus. Per-file COPY (this Dockerfile never uses
154
  # `COPY . .`); without it the lazy import is a no-op and receipts never publish.
155
  COPY szl_corpus_publish.py ./
156
+ # NEMOTRON SIGNED-TRAJECTORY build (2026-06-14): DSSE-signed agent-trajectory
157
+ # corpus pipeline (SZL-Nemo). Honest: DATASET property, not a model claim;
158
+ # QLoRA-ready, training = ROADMAP (2x80GB GPU). nvidia/Nemotron-Agentic-v1
159
+ # mapped under CC BY 4.0 attribution. Served at /signed-corpus.
160
+ COPY szl_trajectory_sign.py szl_nemotron_ingest.py szl_nemotron_corpus.py szl_nemo_verify.py ./
161
 
162
  # Copy serve orchestrator and gates manifest
163
  # ADDITIVE (live-ops): orchestration + AI-observability module — per-file COPY Dockerfile
szl_trajectory_sign.py ADDED
@@ -0,0 +1,310 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # © 2026 Lutar, Stephen P. — SZL Holdings · ORCID 0009-0001-0110-4173 · Doctrine v11/v12
3
+ # Authored by the NEMOTRON SIGNED-TRAJECTORY build team. Co-Authored-By: Perplexity Computer Agent.
4
+ """
5
+ szl_trajectory_sign — DSSE-signed agent-trajectory receipts (SZL-Nemo, NOW slice).
6
+
7
+ WHAT THIS IS (honest framing — read before extending):
8
+ This module instruments the EXISTING SZL agent loop (ReAct + Reflexion +
9
+ Restraint + Auto-Review) to emit a DSSE-SIGNED JSONL receipt per step. Each
10
+ JSONL line is a self-contained, cryptographically-attested record of one
11
+ agent step: {step, role, action, observation, restraint_verdict, ...,
12
+ signature}. The collection of these lines is a "signed agent-trajectory
13
+ corpus" that is QLoRA-ready and verifiable by anyone via a signature check.
14
+
15
+ WHAT THIS IS *NOT* (never claim otherwise — Doctrine honesty gates):
16
+ - This is a DATASET property (signed, provenance-attested trajectories),
17
+ NOT a model claim. No model is trained here.
18
+ - We did NOT reproduce Nemotron Ultra and did NOT train from scratch. The
19
+ full Ultra reproduction is impossible from open artifacts (intermediate
20
+ MOPD teacher checkpoints were never released — see NVIDIA MOPD docs).
21
+ - Actual QLoRA / GRPO training needs >=2x80GB GPUs and is ROADMAP (the Forge
22
+ order, FORGE_NEMOTRON_TRAIN.md). This module runs on CPU only.
23
+ - SZL-Nemo (the future student) = a GOVERNED fine-tune of Qwen3-32B (Apache);
24
+ it is not Ultra and not from-scratch.
25
+
26
+ SIGNING:
27
+ Reuses szl_dsse.sign_payload (ECDSA-P256-SHA256 over the DSSE PAE, backed by
28
+ the SZLHOLDINGS Cosign keypair). When the SZL_COSIGN_PRIVATE_KEY_PEM runtime
29
+ secret is absent the receipt is emitted as an HONEST UNSIGNED envelope
30
+ (signatures: [], honesty marker) — a signature is NEVER fabricated. The
31
+ public key (cosign.pub) is embedded in szl_dsse for offline verification.
32
+
33
+ SCHEMA (one JSONL line per step):
34
+ {
35
+ "schema": "szl.nemo.trajectory.step/v1",
36
+ "trajectory_id": "<uuid4>",
37
+ "step": <int>, # 0-based step index
38
+ "role": "assistant" | "tool" | "user",
39
+ "pattern": "ReAct"|"Reflexion"|"Restraint"|"AutoReview",
40
+ "action": {...} | str, # the agent's action / tool call
41
+ "observation": str, # tool result / environment feedback
42
+ "restraint_verdict": "ALLOW"|"HOLD"|"MONITOR"|"DECLINE",
43
+ "is_correction": bool, # Reflexion backtrack?
44
+ "correction_of": <int> | null,
45
+ "step_hash": "sha256:<hex>", # sha256 over canonical(step content)
46
+ "signature": <DSSE envelope dict>, # signed | honest-unsigned
47
+ "agent_id": "szl-nemo-trajectory-v1",
48
+ "timestamp_utc": "<ISO8601>"
49
+ }
50
+ ADDITIVE · stdlib + szl_dsse only · pure-ish (signing reads a runtime secret).
51
+ """
52
+ from __future__ import annotations
53
+
54
+ import hashlib
55
+ import json
56
+ import uuid
57
+ from datetime import datetime, timezone
58
+ from typing import Any, Dict, List, Optional
59
+
60
+ STEP_SCHEMA = "szl.nemo.trajectory.step/v1"
61
+ TRAJ_SCHEMA = "szl.nemo.trajectory/v1"
62
+ STEP_PAYLOAD_TYPE = "application/vnd.szl.nemo.trajectory.step+json"
63
+ AGENT_ID = "szl-nemo-trajectory-v1"
64
+
65
+ # Canonical SZL agent patterns (the four loops we instrument).
66
+ PATTERNS = ("ReAct", "Reflexion", "Restraint", "AutoReview")
67
+ # Restraint verdicts — "never engage on doubt": HOLD/MONITOR are the safe defaults.
68
+ VERDICTS = ("ALLOW", "HOLD", "MONITOR", "DECLINE")
69
+
70
+
71
+ def _utcnow_iso() -> str:
72
+ return datetime.now(timezone.utc).isoformat()
73
+
74
+
75
+ def _canon(obj: Any) -> bytes:
76
+ return json.dumps(obj, sort_keys=True, separators=(",", ":"),
77
+ ensure_ascii=False).encode("utf-8")
78
+
79
+
80
+ def step_hash(action: Any, observation: Any, restraint_verdict: str) -> str:
81
+ """Deterministic SHA-256 over the step's signable content."""
82
+ body = _canon({"action": action, "observation": observation,
83
+ "restraint_verdict": restraint_verdict})
84
+ return "sha256:" + hashlib.sha256(body).hexdigest()
85
+
86
+
87
+ def new_trajectory_id() -> str:
88
+ return str(uuid.uuid4())
89
+
90
+
91
+ # --------------------------------------------------------------------------- #
92
+ # Signing
93
+ # --------------------------------------------------------------------------- #
94
+ def _sign(payload: Dict[str, Any]) -> Dict[str, Any]:
95
+ """DSSE-sign a step payload; honest-unsigned fallback if no key. Never raises."""
96
+ try:
97
+ import szl_dsse
98
+ return szl_dsse.sign_payload(payload, STEP_PAYLOAD_TYPE)
99
+ except Exception as exc: # honest degrade, never fabricate
100
+ return {
101
+ "payloadType": STEP_PAYLOAD_TYPE,
102
+ "signatures": [],
103
+ "signed": False,
104
+ "honesty": f"UNSIGNED — signer module unavailable ({type(exc).__name__}); "
105
+ "no signature fabricated.",
106
+ }
107
+
108
+
109
+ def signing_available() -> bool:
110
+ try:
111
+ import szl_dsse
112
+ return bool(szl_dsse.signing_available())
113
+ except Exception:
114
+ return False
115
+
116
+
117
+ # --------------------------------------------------------------------------- #
118
+ # Per-step receipt builder
119
+ # --------------------------------------------------------------------------- #
120
+ def sign_step(
121
+ *,
122
+ trajectory_id: str,
123
+ step: int,
124
+ action: Any,
125
+ observation: Any = "",
126
+ role: str = "assistant",
127
+ pattern: str = "ReAct",
128
+ restraint_verdict: str = "ALLOW",
129
+ is_correction: bool = False,
130
+ correction_of: Optional[int] = None,
131
+ tool_calls: Optional[List[Dict[str, Any]]] = None,
132
+ extra: Optional[Dict[str, Any]] = None,
133
+ ) -> Dict[str, Any]:
134
+ """Build ONE DSSE-signed trajectory-step receipt (the JSONL line dict).
135
+
136
+ The signature commits to the {action, observation, restraint_verdict} content
137
+ via the canonical step payload, so any third party can re-derive the PAE and
138
+ verify it against the published cosign.pub.
139
+ """
140
+ if pattern not in PATTERNS:
141
+ pattern = "ReAct"
142
+ if restraint_verdict not in VERDICTS:
143
+ restraint_verdict = "MONITOR" # honest default: never assume ALLOW on doubt
144
+ sh = step_hash(action, observation, restraint_verdict)
145
+ # The signed payload is the content-bearing core (stable, dedup-friendly).
146
+ payload = {
147
+ "schema": STEP_SCHEMA,
148
+ "trajectory_id": trajectory_id,
149
+ "step": int(step),
150
+ "role": role,
151
+ "pattern": pattern,
152
+ "action": action,
153
+ "observation": observation,
154
+ "restraint_verdict": restraint_verdict,
155
+ "is_correction": bool(is_correction),
156
+ "correction_of": correction_of,
157
+ "tool_calls": tool_calls or [],
158
+ "step_hash": sh,
159
+ "agent_id": AGENT_ID,
160
+ "timestamp_utc": _utcnow_iso(),
161
+ }
162
+ if extra:
163
+ payload["extra"] = extra
164
+ envelope = _sign(payload)
165
+ line = dict(payload)
166
+ line["signature"] = envelope
167
+ return line
168
+
169
+
170
+ # --------------------------------------------------------------------------- #
171
+ # Trajectory recorder — wraps an agent run, emits signed JSONL
172
+ # --------------------------------------------------------------------------- #
173
+ class SignedTrajectory:
174
+ """Accumulate DSSE-signed step receipts for one agent run, then seal.
175
+
176
+ Usage (instrumentation point — wrap the existing loop's step()):
177
+ t = SignedTrajectory(task="counter-UAS track+classify", environment="cuas")
178
+ t.add(action="track_air_vehicle(...)", observation="3 tracks", pattern="ReAct")
179
+ t.add(action="classify_threat(...)", observation="ERROR: no IFF",
180
+ restraint_verdict="HOLD", pattern="Restraint")
181
+ t.add(action="retry classify with EO/IR", observation="hostile UAS",
182
+ is_correction=True, correction_of=1, pattern="Reflexion")
183
+ sealed = t.seal(outcome="success") # provenance block + JSONL
184
+ """
185
+
186
+ def __init__(self, *, task: str = "", environment: str = "szl",
187
+ trajectory_id: Optional[str] = None,
188
+ label: str = "SAMPLE"):
189
+ self.trajectory_id = trajectory_id or new_trajectory_id()
190
+ self.task = task
191
+ self.environment = environment
192
+ self.label = label # LIVE | SAMPLE | MODELED — honesty surface
193
+ self.steps: List[Dict[str, Any]] = []
194
+
195
+ def add(self, **kwargs: Any) -> Dict[str, Any]:
196
+ step = len(self.steps)
197
+ line = sign_step(trajectory_id=self.trajectory_id, step=step, **kwargs)
198
+ self.steps.append(line)
199
+ return line
200
+
201
+ def jsonl(self) -> str:
202
+ return "\n".join(json.dumps(s, ensure_ascii=False) for s in self.steps)
203
+
204
+ def provenance(self, outcome: str = "unknown") -> Dict[str, Any]:
205
+ corrections = sum(1 for s in self.steps if s.get("is_correction"))
206
+ verdicts = [s.get("restraint_verdict") for s in self.steps]
207
+ signed = sum(1 for s in self.steps
208
+ if (s.get("signature", {}) or {}).get("signed"))
209
+ return {
210
+ "schema": TRAJ_SCHEMA,
211
+ "trajectory_id": self.trajectory_id,
212
+ "task": self.task,
213
+ "environment": self.environment,
214
+ "label": self.label,
215
+ "signer": AGENT_ID,
216
+ "signed_at": _utcnow_iso(),
217
+ "total_steps": len(self.steps),
218
+ "corrections": corrections,
219
+ "verdicts": verdicts,
220
+ "signed_steps": signed,
221
+ "all_signed": signed == len(self.steps) and len(self.steps) > 0,
222
+ "outcome": outcome,
223
+ "source": "szl",
224
+ "verified": False, # signatures present; independent verify is the user's job
225
+ "honesty": (
226
+ "DSSE-signed agent-trajectory corpus (DATASET property, not a model "
227
+ "claim). QLoRA-ready; actual training = ROADMAP (needs 2x80GB GPU). "
228
+ "Not an Ultra reproduction; not trained from scratch."
229
+ ),
230
+ }
231
+
232
+ def seal(self, outcome: str = "unknown") -> Dict[str, Any]:
233
+ return {"provenance": self.provenance(outcome), "steps": self.steps,
234
+ "jsonl": self.jsonl()}
235
+
236
+
237
+ # --------------------------------------------------------------------------- #
238
+ # Verifier — anyone can run this to check every step signature
239
+ # --------------------------------------------------------------------------- #
240
+ def verify_step(line: Dict[str, Any]) -> Dict[str, Any]:
241
+ """Verify ONE signed-step JSONL line: re-derive step_hash AND verify the DSSE
242
+ signature against the embedded payload. Returns a structured verdict."""
243
+ out: Dict[str, Any] = {"trajectory_id": line.get("trajectory_id"),
244
+ "step": line.get("step")}
245
+ # 1) content integrity: recompute step_hash
246
+ recomputed = step_hash(line.get("action"), line.get("observation", ""),
247
+ line.get("restraint_verdict", "ALLOW"))
248
+ out["hash_ok"] = (recomputed == line.get("step_hash"))
249
+ # 2) signature: verify the DSSE envelope (if signed)
250
+ env = line.get("signature") or {}
251
+ sigs = env.get("signatures") or []
252
+ if not sigs:
253
+ out["signed"] = False
254
+ out["sig_ok"] = False
255
+ out["reason"] = env.get("honesty", "unsigned")
256
+ return out
257
+ out["signed"] = True
258
+ try:
259
+ import szl_dsse
260
+ verdict = szl_dsse.verify_envelope(env)
261
+ out["sig_ok"] = bool(verdict.get("verified"))
262
+ out["sig_detail"] = verdict
263
+ except Exception as exc:
264
+ out["sig_ok"] = False
265
+ out["reason"] = f"{type(exc).__name__}: {exc}"
266
+ return out
267
+
268
+
269
+ def verify_jsonl(text: str) -> Dict[str, Any]:
270
+ """Verify every step in a JSONL corpus blob. Returns aggregate stats."""
271
+ lines = [ln for ln in text.splitlines() if ln.strip()]
272
+ results = []
273
+ for ln in lines:
274
+ try:
275
+ results.append(verify_step(json.loads(ln)))
276
+ except Exception as exc:
277
+ results.append({"parse_error": f"{type(exc).__name__}: {exc}"})
278
+ total = len(results)
279
+ hash_ok = sum(1 for r in results if r.get("hash_ok"))
280
+ signed = sum(1 for r in results if r.get("signed"))
281
+ sig_ok = sum(1 for r in results if r.get("sig_ok"))
282
+ return {
283
+ "total_steps": total,
284
+ "hash_ok": hash_ok,
285
+ "signed": signed,
286
+ "sig_ok": sig_ok,
287
+ "all_hash_ok": hash_ok == total and total > 0,
288
+ "all_sig_ok": sig_ok == total and total > 0,
289
+ "results": results,
290
+ }
291
+
292
+
293
+ # --------------------------------------------------------------------------- #
294
+ # Self-check (pure)
295
+ # --------------------------------------------------------------------------- #
296
+ if __name__ == "__main__":
297
+ t = SignedTrajectory(task="demo self-check", environment="selftest")
298
+ t.add(action="step A", observation="ok", pattern="ReAct")
299
+ t.add(action="step B", observation="ERROR", restraint_verdict="HOLD",
300
+ pattern="Restraint")
301
+ t.add(action="retry B", observation="ok", is_correction=True,
302
+ correction_of=1, pattern="Reflexion")
303
+ sealed = t.seal(outcome="success")
304
+ v = verify_jsonl(sealed["jsonl"])
305
+ assert v["total_steps"] == 3
306
+ assert v["all_hash_ok"], v
307
+ print(json.dumps({"signing_available": signing_available(),
308
+ "provenance": sealed["provenance"], "verify": {
309
+ k: v[k] for k in ("total_steps", "all_hash_ok",
310
+ "signed", "sig_ok")}}, indent=2))