"""Evaluate a stopped campaign's latest checkpoint on the fixed validation set. Example: python scripts/final_validation.py --checkpoint /run/training/checkpoint.pt --reference-predictions /fixed/validation.json --reference-sha256 SHA256 --evidence-dir /parent/artifacts --output /new/final-validation --deadline 2026-09-17T09:00:00Z Original checkpoints remain untouched. The output best.pt is fleet-compatible; selection uses only the existing 512 validation decisions and frozen policy. """ import argparse from datetime import datetime, timezone import fcntl import hashlib import json import math import os from pathlib import Path import shutil import signal import subprocess import sys import time ROOT = Path(__file__).resolve().parents[1] sys.path.insert(0, str(ROOT)) import torch import experiment from scripts.fleet_campaign import (checksum, identity, timestamp, validation_score, verify_candidate, verify_compatibility, verify_correctness) from selection import SELECTION_METRIC, validation_selection MINIMUM_IMPROVEMENT = 0.001 STRIPPED_KEYS = ('optimizer', 'random_state', 'torch_rng', 'cuda_rng') def utc(): return datetime.now(timezone.utc).isoformat() def read_json(path): return json.loads(Path(path).read_text()) def stable_copy(source, destination, expected=None): before = checksum(source) if expected is not None and before != expected: raise ValueError('Source checksum differs from expected artifact') destination.parent.mkdir(parents=True, exist_ok=True) shutil.copy2(source, destination) if checksum(source) != before or checksum(destination) != before: raise ValueError('Source changed while freezing evidence') return before def source_proof(directory, saved): manifest = read_json(directory / 'manifest.json') if manifest['git_commit'] != saved['source_commit']: raise ValueError('Latest checkpoint and training manifest source differ') if any(manifest[key] != saved[key] for key in ('config', 'data_signature', 'model_provenance')): raise ValueError('Latest checkpoint and training manifest configuration/provenance differ') pid = manifest.get('pid') proc = Path('/proc') / str(pid) / 'cmdline' if proc.exists(): command = proc.read_bytes().decode().split('\0') if any('experiment.py' in arg for arg in command) and manifest['config'].get('output') in command: raise ValueError('Source trainer is still running') if not manifest.get('source_sha256'): raise ValueError('Training source hash inventory missing') for name, expected in manifest['source_sha256'].items(): if Path(name).is_absolute() or '..' in Path(name).parts: raise ValueError('Unsafe training source path') content = subprocess.check_output(['git', 'show', saved['source_commit'] + ':' + name], cwd=ROOT, stderr=subprocess.DEVNULL, timeout=30) if hashlib.sha256(content).hexdigest() != expected: raise ValueError('Training source hash differs from recorded commit') for name in ('training_model.py', 'decision_model.py', 'selection.py'): if checksum(ROOT / name) != manifest['source_sha256'][name]: raise ValueError('Scoring implementation changed since training') return manifest def choose_latest(previous_score, latest_score): if not all(isinstance(v, (int, float)) and math.isfinite(v) for v in (previous_score, latest_score)): raise ValueError('Selection scores must be finite') return latest_score < previous_score - MINIMUM_IMPROVEMENT def validation_rows(dataset, reference, scorer): # Other split bytes are hashed by verify_compatibility; their labels are never parsed. rows = [json.loads(line) for line in (Path(dataset) / 'validation.jsonl').read_text().splitlines()] if len(rows) != 512 or identity(rows) != identity(reference): raise ValueError('Dataset does not contain exactly the frozen 512 validation decisions') by_id = {row['id']: row for row in rows} ordered = [by_id[row['id']] for row in reference] for row in ordered: row['_sequences'] = scorer.sequences(row) # Oversized rows fail; none are dropped/truncated. return ordered def run(args): Path('/proc/self/oom_score_adj').write_text('0') experiment.STOP = False deadline = timestamp(args.deadline) if time.time() >= deadline: raise ValueError('Final validation deadline already elapsed') out = Path(args.output).resolve() out.mkdir(parents=True, exist_ok=False) state = {'status': 'running', 'stage': 'freezing_sources', 'pid': os.getpid(), 'command': [sys.executable, *sys.argv], 'started_utc': utc(), 'deadline': args.deadline, 'helper_sha256': checksum(__file__), 'helper_source_commit': subprocess.check_output( ['git', 'rev-parse', 'HEAD'], cwd=ROOT, text=True).strip(), 'reserved_predictions_accessed': False} def update(stage, **values): state.update(stage=stage, heartbeat_utc=utc(), **values) experiment.write_json(out / 'state.json', state) update('freezing_sources') try: source_checkpoint = Path(args.checkpoint).resolve() source = source_checkpoint.parent reference_path = Path(args.reference_predictions).resolve() if checksum(reference_path) != args.reference_sha256: raise ValueError('Frozen validation reference checksum mismatch') reference = read_json(reference_path) original_sha = stable_copy(source_checkpoint, out / 'source/checkpoint.pt') previous_sha = stable_copy(source / 'best.pt', out / 'source/best.pt') stable_copy(reference_path, out / 'reference_predictions.json', args.reference_sha256) loaded = torch.load(out / 'source/checkpoint.pt', map_location='cpu', weights_only=False) latest = {key: value for key, value in loaded.items() if key not in STRIPPED_KEYS} del loaded previous = torch.load(out / 'source/best.pt', map_location='cpu', weights_only=False) if any(saved.get('format') != 'opensysone-adapter-v1' or 'temperature' in saved for saved in (latest, previous)): raise ValueError('Final validation requires raw trained artifacts') if latest.get('selection_metric') != SELECTION_METRIC or previous.get('selection_metric') != SELECTION_METRIC: raise ValueError('Final validation cannot change the frozen selection policy') manifest = source_proof(source, latest) stable_copy(source / 'manifest.json', out / 'source/manifest.json') for saved in (latest, previous): verify_compatibility(saved) if not saved.get('trainable_state') or any(not bool(torch.isfinite(value).all()) for value in saved['trainable_state'].values()): raise ValueError('Checkpoint trainable tensors are missing/nonfinite') if (latest['data_signature'] != previous['data_signature'] or latest['model_provenance'] != previous['model_provenance'] or any(latest['config'][key] != previous['config'][key] for key in ('rank', 'alpha', 'adapters', 'max_tokens', 'branch_batch_size'))): raise ValueError('Latest and selected checkpoints use incompatible scoring configurations') model_path = Path(latest['config']['model']) if read_json(model_path / 'opensysone-provenance.json') != latest['model_provenance']: raise ValueError('Available base model pin differs from training artifact') directories = [source, *[Path(path).resolve() for path in args.evidence_dir]] if previous['config'].get('output'): directories.append(Path(previous['config']['output'])) previous_predictions, previous_metrics, previous_evidence = None, None, None for directory in directories: for name in (f"validation_step_{previous['step']:06d}_predictions.json", 'best_validation_predictions.json', 'initial_validation_predictions.json' if previous['step'] == 0 else ''): path = directory / name if not name or not path.is_file(): continue predictions = read_json(path) try: metrics = verify_candidate(previous, predictions, reference) except (KeyError, TypeError, ValueError): continue previous_predictions, previous_metrics, previous_evidence = predictions, metrics, path break if previous_evidence: break if previous_evidence is None: raise ValueError('Original durable best lacks matching fixed-validation evidence') stable_copy(previous_evidence, out / 'source/best_validation_predictions.json') source_correctness = verify_correctness(source) stable_copy(source_correctness, out / 'source/correctness.json') latest['final_validation'] = {'original_checkpoint_sha256': original_sha, 'source_checkpoint': str(source_checkpoint), 'helper_sha256': state['helper_sha256'], 'helper_source_commit': state['helper_source_commit']} experiment.save_torch(out / 'latest.pt', latest) # Durable reconstruction before any GPU evaluation. update('loading', latest_step=latest['step'], previous_best_step=previous['step'], source_checkpoint_sha256=original_sha, source_best_sha256=previous_sha) experiment.guard_memory(args.device) scorer, reconstructed = experiment.load_artifact(out / 'latest.pt', device=args.device) del reconstructed if scorer.provenance != latest['model_provenance']: raise ValueError('Reconstructed base model provenance mismatch') rows = validation_rows(latest['config']['dataset'], reference, scorer) update('correctness') checks = experiment.correctness(scorer, rows, allow_stop=True) experiment.write_json(out / 'latest_correctness.json', checks) update('validating', completed_decisions=0) predictions = [] started = time.monotonic() for start in range(0, len(rows), 128): predictions.extend(experiment.predict(scorer, rows[start:start+128], deadline=deadline)) update('validating', completed_decisions=len(predictions)) elapsed = time.monotonic() - started raw = validation_score(predictions, reference) selected_score = validation_selection(predictions) latest.update(best_validation_macro_nll=raw['macro_nll'], selection_metric=SELECTION_METRIC, best_validation_selection_score=selected_score['score'], validation_selection=selected_score) verify_candidate(latest, predictions, reference) experiment.save_torch(out / 'latest.evaluated.pt', latest) experiment.write_json(out / 'latest_validation_predictions.json', predictions) experiment.write_json(out / f"validation_step_{latest['step']:06d}_predictions.json", predictions) experiment.write_json(out / 'latest_validation_selection.json', selected_score) improved = choose_latest(previous_metrics['selection_score'], selected_score['score']) if checksum(source_checkpoint) != original_sha or checksum(source / 'best.pt') != previous_sha: raise ValueError('Source artifacts changed during final validation') selected, selected_predictions = (latest, predictions) if improved else (previous, previous_predictions) if improved: stable_copy(out / 'latest.evaluated.pt', out / 'best.pt') experiment.write_json(out / 'correctness_final.json', checks) else: stable_copy(out / 'source/best.pt', out / 'best.pt', previous_sha) stable_copy(out / 'source/correctness.json', out / 'correctness_final.json') experiment.write_json(out / 'best_validation_predictions.json', selected_predictions) experiment.write_json(out / f"validation_step_{selected['step']:06d}_predictions.json", selected_predictions) if selected['step'] == 0: experiment.write_json(out / 'initial_validation_predictions.json', selected_predictions) selected_metrics = verify_candidate(selected, selected_predictions, reference) experiment.write_json(out / 'best_validation_selection.json', selected_metrics['selection']) experiment.write_json(out / 'manifest.json', {**manifest, 'config': selected['config'], 'final_validation': {**state, 'source_manifest': str(source / 'manifest.json'), 'selected_training_source_commit': selected['source_commit']}}) result = {'status': 'complete', 'latest_step': latest['step'], 'previous_best_step': previous['step'], 'selected_step': selected['step'], 'promoted_latest': improved, 'minimum_improvement': MINIMUM_IMPROVEMENT, 'previous_selection_score': previous_metrics['selection_score'], 'latest_selection_score': selected_score['score'], 'latest_raw_macro_nll': raw['macro_nll'], 'latest_accuracy': raw['accuracy'], 'selected_selection_score': selected_metrics['selection_score'], 'selected_accuracy': selected_metrics['accuracy'], 'validation_count': 512, 'validation_seconds': elapsed, 'selection_metric': SELECTION_METRIC, 'reference_sha256': args.reference_sha256, 'source_checkpoint_sha256': original_sha, 'source_best_sha256': previous_sha, 'best_sha256': checksum(out / 'best.pt'), 'latest_sha256': checksum(out / 'latest.evaluated.pt'), 'reserved_predictions_accessed': False, 'optimizer_restored': False, 'optimizer_updates': 0, 'completed_utc': utc()} if checksum(source_checkpoint) != original_sha or checksum(source / 'best.pt') != previous_sha: raise ValueError('Source artifacts changed during final validation') experiment.write_json(out / 'final-validation.json', result) experiment.write_json(out / 'summary.json', result) update('complete', status='complete', selected_step=selected['step'], promoted_latest=improved) return result except BaseException as error: update('failed', status='failed', error_type=type(error).__name__) raise def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument('--checkpoint', required=True) parser.add_argument('--reference-predictions', required=True) parser.add_argument('--reference-sha256', required=True) parser.add_argument('--evidence-dir', action='append', default=[]) parser.add_argument('--output', required=True) parser.add_argument('--deadline', required=True) parser.add_argument('--device', choices=('cuda', 'cpu'), default='cuda') args = parser.parse_args() lock_path = Path.home() / 'ai/opensysone/runs/.smoke.lock' with lock_path.open('w') as lock: fcntl.flock(lock, fcntl.LOCK_EX | fcntl.LOCK_NB) signal.signal(signal.SIGTERM, experiment.request_stop) signal.signal(signal.SIGINT, experiment.request_stop) def expired(*unused): raise TimeoutError('Final validation deadline reached') signal.signal(signal.SIGALRM, expired) signal.setitimer(signal.ITIMER_REAL, max(0.001, timestamp(args.deadline) - time.time())) try: result = run(args) finally: signal.setitimer(signal.ITIMER_REAL, 0) print(json.dumps(result), flush=True) if __name__ == '__main__': main()