Download source/scripts/verify_artifact.py from andyshu/opensysone: direct link, hf CLI and curl.
- Browser
- Download file 9.97 kB
-
https://huggingface.co/andyshu/opensysone/resolve/main/source/scripts/verify_artifact.py
- Command line
-
hf download hf://andyshu/opensysone/source/scripts/verify_artifact.py
-
curl -L -o verify_artifact.py https://huggingface.co/andyshu/opensysone/resolve/main/source/scripts/verify_artifact.py
9.97 kB
| """Verify a trusted trained checkpoint, longest training inputs and real HTTP. | |
| No optimizer step is taken and the source checkpoint is never changed. Optimizer | |
| state is restored so the stress pass measures the memory needed during training. | |
| """ | |
| import argparse | |
| from datetime import datetime, timezone | |
| import fcntl | |
| import json | |
| import os | |
| from pathlib import Path | |
| import sys | |
| import threading | |
| import time | |
| sys.path.insert(0, str(Path(__file__).resolve().parents[1])) | |
| import torch | |
| from experiment import (data_for, decision_backward, guard_memory, load_artifact, | |
| predict, sha256, write_json) | |
| from jev_harness import LocalBackend, RemoteBackend, compile_request, create_server | |
| def main(): | |
| parser = argparse.ArgumentParser(description=__doc__) | |
| parser.add_argument('--checkpoint', required=True) | |
| parser.add_argument('--reference', required=True) | |
| parser.add_argument('--output', required=True) | |
| parser.add_argument('--inference-max-tokens',type=int,default=1024) | |
| args = parser.parse_args() | |
| lock_path = Path.home()/'ai/opensysone/runs/.smoke.lock' | |
| lock = lock_path.open('w') | |
| try: | |
| fcntl.flock(lock,fcntl.LOCK_EX | fcntl.LOCK_NB) | |
| except BlockingIOError: | |
| raise RuntimeError('Another project run holds the model-load lock') from None | |
| out = Path(args.output).resolve() | |
| out.mkdir(parents=True, exist_ok=False) | |
| before = sha256(args.checkpoint) | |
| write_json(out/'manifest.json',{'pid':os.getpid(),'checkpoint_sha256':before, | |
| 'started_utc':datetime.now(timezone.utc).isoformat(), | |
| 'verification_source_sha256':sha256(__file__)}) | |
| guard_memory() | |
| scorer, artifact = load_artifact(args.checkpoint) | |
| if not 256 <= args.inference_max_tokens <= scorer.lm.config.max_position_embeddings: | |
| raise ValueError('Verification inference limit must be 256 to the base context limit') | |
| config = artifact['config'] | |
| data, signature = data_for(scorer, config['dataset'], out) | |
| if signature != artifact['data_signature']: | |
| raise ValueError('Frozen data/model signature mismatch') | |
| reference = json.loads(Path(args.reference).read_text()) | |
| by_id = {row['id']: row for row in reference} | |
| selected = [] | |
| for family in sorted({r['family'] for r in data['validation']}): | |
| selected.extend([r for r in data['validation'] if r['family'] == family][:4]) | |
| fresh = predict(scorer, selected) | |
| max_difference = max(abs(a-b) for row in fresh | |
| for a,b in zip(row['probabilities'], by_id[row['id']]['probabilities'])) | |
| if max_difference > 1e-4: | |
| raise RuntimeError(f'Fresh reconstruction differs: {max_difference}') | |
| write_json(out / 'reload_predictions.json', fresh) | |
| parameters = [p for p in scorer.parameters() if p.requires_grad] | |
| optimizer = torch.optim.AdamW([ | |
| {'params': [p for p in scorer.lm.parameters() if p.requires_grad], 'lr': config['lr']}, | |
| {'params': scorer.head.parameters(), 'lr': config['head_lr']}], weight_decay=0.01) | |
| optimizer.load_state_dict(artifact['optimizer']) | |
| # Longest complete decision in each family, plus each candidate-count class. | |
| stress_rows = {} | |
| for key in sorted({r['family'] for r in data['train']}): | |
| row = max((r for r in data['train'] if r['family']==key), | |
| key=lambda r:max(map(len,r['_sequences']))) | |
| stress_rows[row['id']] = row | |
| for count in sorted({len(r['choices']) for r in data['train']}): | |
| row = max((r for r in data['train'] if len(r['choices'])==count), | |
| key=lambda r:max(map(len,r['_sequences']))) | |
| stress_rows[row['id']] = row | |
| stress = [] | |
| for row in stress_rows.values(): | |
| scorer.train() | |
| optimizer.zero_grad(set_to_none=True) | |
| loss = decision_backward(scorer,row,1,config.get('two_pass',False)) | |
| norm = torch.nn.utils.clip_grad_norm_(parameters,1.0,error_if_nonfinite=True) | |
| stress.append({'id':row['id'],'family':row['family'],'branches':len(row['choices']), | |
| 'max_branch_tokens':max(map(len,row['_sequences'])), | |
| 'loss':loss,'gradient_norm':norm.item(), | |
| 'peak_cuda_allocated_bytes':torch.cuda.max_memory_allocated()}) | |
| write_json(out / 'stress.json', stress) | |
| optimizer.zero_grad(set_to_none=True) | |
| training_peak = torch.cuda.max_memory_allocated() | |
| training_reserved = torch.cuda.max_memory_reserved() | |
| del optimizer | |
| torch.cuda.reset_peak_memory_stats() | |
| scorer.eval() | |
| backend = LocalBackend.__new__(LocalBackend) | |
| backend.scorer, backend.temperature = scorer, artifact.get('temperature',1.0) | |
| backend.checkpoint, backend.calibrated = str(Path(args.checkpoint).resolve()), 'temperature' in artifact | |
| backend.max_tokens = scorer.max_tokens | |
| backend.model_name = 'opensysone-'+artifact['model_provenance']['model_id'].split('/')[-1].lower() | |
| payload = json.loads((Path(__file__).resolve().parents[1]/'examples/jev_request.json').read_text()) | |
| direct = backend(payload) | |
| server = create_server(backend,port=0,api_key='local-verification-only') | |
| worker = threading.Thread(target=server.serve_forever,daemon=True) | |
| worker.start() | |
| try: | |
| remote = RemoteBackend(f'http://127.0.0.1:{server.server_address[1]}', | |
| api_key='local-verification-only',attempts=1) | |
| tick = time.perf_counter() | |
| response = remote(payload) | |
| elapsed = time.perf_counter()-tick | |
| if response != direct: | |
| raise RuntimeError('HTTP response differs from direct trained-model inference') | |
| try: | |
| RemoteBackend(remote.base_url,api_key='wrong-key',attempts=1)(payload) | |
| except RuntimeError as error: | |
| if str(error) != 'Jev HTTP 401; request failed': | |
| raise | |
| else: | |
| raise RuntimeError('HTTP authentication did not reject an invalid key') | |
| try: | |
| remote({**payload,'state':'oversized '* (config['max_tokens']*2)}) | |
| except RuntimeError as error: | |
| if str(error) != 'Jev HTTP 422; request failed': | |
| raise | |
| else: | |
| raise RuntimeError('HTTP did not reject an oversized complete candidate') | |
| # Inference has no gradient graphs or optimizer-step temporaries. Verify | |
| # a useful larger API context separately from the training token limit. | |
| backend.scorer.max_tokens = backend.max_tokens = args.inference_max_tokens | |
| for words in range(args.inference_max_tokens,32,-8): | |
| long_payload = {'model':'opensysone','state':'background '*words+'Please help today.', | |
| 'questions':{'urgent':{'type':'noul','instructions':'Does the customer express urgency?'}}} | |
| try: | |
| long_sequences = scorer.sequences(compile_request(long_payload)[0][4]) | |
| break | |
| except ValueError as error: | |
| if 'no truncation' not in str(error): | |
| raise | |
| else: | |
| raise RuntimeError('Could not construct a bounded long-context request') | |
| if max(map(len,long_sequences)) < args.inference_max_tokens-24: | |
| raise RuntimeError('Long-context probe did not reach its intended token length') | |
| long_direct = backend(long_payload) | |
| long_response = remote(long_payload) | |
| if long_response != long_direct: | |
| raise RuntimeError('Long-context HTTP output differs from direct inference') | |
| write_json(out/'http_long_context_response.json',long_response) | |
| large_payload = {'model':'opensysone','state':'A customer needs help with a failed payment.', | |
| 'questions':{'routing':{'type':'choice', | |
| 'instructions':'Which candidate routing category best fits the request?', | |
| 'criteria':{f'category_{i}':f'Routing category {i}' for i in range(255)}}}} | |
| remote.timeout = 300 | |
| tick = time.perf_counter() | |
| large_response = remote(large_payload) | |
| large_elapsed = time.perf_counter()-tick | |
| if len(large_response['answers']['routing']['probabilities']) != 255: | |
| raise RuntimeError('HTTP maximum-choice request lost alternatives') | |
| write_json(out/'http_255_choices_response.json',large_response) | |
| finally: | |
| server.shutdown() | |
| server.server_close() | |
| worker.join(timeout=5) | |
| write_json(out / 'http_response.json', response) | |
| if sha256(args.checkpoint) != before: | |
| raise RuntimeError('Verification changed the source checkpoint') | |
| result = {'status':'passed','checkpoint':backend.checkpoint,'checkpoint_sha256':before, | |
| 'step':artifact['step'],'fresh_reload_decisions':len(fresh), | |
| 'fresh_reload_probability_max_abs':max_difference, | |
| 'optimizer_state_restored':True,'optimizer_steps_taken':0, | |
| 'longest_input_stress':stress,'http_model':backend.model_name, | |
| 'http_matches_direct':True,'http_invalid_key_status':401, | |
| 'http_oversized_input_status':422,'http_255_choices':True, | |
| 'inference_max_tokens':args.inference_max_tokens,'http_long_context_branch_tokens':list(map(len,long_sequences)), | |
| 'http_long_context_matches_direct':True, | |
| 'http_255_choices_seconds':large_elapsed, | |
| 'http_end_to_end_seconds':elapsed,'temperature_fitted':backend.calibrated, | |
| 'training_peak_cuda_allocated_bytes':training_peak, | |
| 'inference_peak_cuda_allocated_bytes':torch.cuda.max_memory_allocated(), | |
| 'peak_cuda_allocated_bytes':max(training_peak,torch.cuda.max_memory_allocated()), | |
| 'peak_cuda_reserved_bytes':max(training_reserved,torch.cuda.max_memory_reserved())} | |
| write_json(out / 'verification.json', result) | |
| print(json.dumps(result),flush=True) | |
| if __name__ == '__main__': | |
| main() | |