Download source/tests/test_expanded_startup.py from andyshu/opensysone: direct link, hf CLI and curl.
- Browser
- Download file 6.65 kB
-
https://huggingface.co/andyshu/opensysone/resolve/f2d6f8daa15bd21c316e61249f45ac16cbb79d45/source/tests/test_expanded_startup.py
- Command line
-
hf download hf://andyshu/opensysone@f2d6f8daa15bd21c316e61249f45ac16cbb79d45/source/tests/test_expanded_startup.py
-
curl -L -o test_expanded_startup.py https://huggingface.co/andyshu/opensysone/resolve/f2d6f8daa15bd21c316e61249f45ac16cbb79d45/source/tests/test_expanded_startup.py
6.65 kB
| import copy | |
| import json | |
| from pathlib import Path | |
| import random | |
| import tempfile | |
| import unittest | |
| import torch | |
| from scripts import verify_expanded_startup as audit | |
| def predictions(): | |
| return [{'id': f'{family}-{i}', 'group': f'{family}-{i}', 'family': family, | |
| 'target': 0, 'choices': ['yes', 'no'], 'logits': [0.0, 0.0], | |
| 'probabilities': [0.5, 0.5], 'log_probabilities': [-0.693147, -0.693147]} | |
| for family in ('arc', 'banking', 'boolq', 'snli') for i in range(128)] | |
| class StartupVerificationTests(unittest.TestCase): | |
| def test_initial_checkpoint_proves_empty_optimizer_and_parent_weights(self): | |
| parent = {'trainable_state': {'head': torch.tensor([1.0])}} | |
| pilot = {'config': {'seed': 433}, 'initialization': {'restores_rng': False}, | |
| 'data_signature': 'expanded', 'source_commit': 'source', 'model_provenance': 'base'} | |
| initial = {**copy.deepcopy(pilot), 'step': 0, 'optimizer': {'state': {}}, | |
| 'trainable_state': copy.deepcopy(parent['trainable_state']), | |
| 'random_state': random.Random(433).getstate(), | |
| 'torch_rng': torch.tensor([1]), 'cuda_rng': [torch.tensor([2])]} | |
| self.assertTrue(audit.verify_initial_state(initial, pilot, parent, torch)['adam_state_empty']) | |
| for kind in ('weights', 'optimizer', 'rng'): | |
| changed = copy.deepcopy(initial) | |
| if kind == 'weights': changed['trainable_state']['head'][0] += 1 | |
| elif kind == 'optimizer': changed['optimizer']['state'][0] = {'step': 1500} | |
| else: changed['random_state'] = random.Random(432).getstate() | |
| with self.subTest(kind=kind), self.assertRaises(ValueError): | |
| audit.verify_initial_state(changed, pilot, parent, torch) | |
| def test_prediction_comparison_uses_identity_not_order(self): | |
| rows = predictions() | |
| proof = audit.prediction_parity(rows, list(reversed(copy.deepcopy(rows)))) | |
| self.assertEqual(proof['rows'], 512) | |
| self.assertTrue(all(v == 0 for v in proof['max_abs_difference'].values())) | |
| def test_inexact_predictions_changed_identity_and_duplicates_rejected(self): | |
| rows = predictions() | |
| for key, value in [('logits', [1e-15, 0.0]), ('target', 1), | |
| ('choices', ['no', 'yes']), ('group', 'other'), | |
| ('id', rows[1]['id']), ('probabilities', [0.5])]: | |
| with self.subTest(key=key): | |
| changed = copy.deepcopy(rows) | |
| changed[0][key] = value | |
| with self.assertRaises(ValueError): | |
| audit.prediction_parity(rows, changed) | |
| def test_prediction_nan_and_partial_cut_rejected(self): | |
| rows = predictions() | |
| with self.assertRaisesRegex(ValueError, '512'): | |
| audit.prediction_parity(rows, rows[:-1]) | |
| rows[0]['logits'][0] = float('nan') | |
| with self.assertRaisesRegex(ValueError, 'Non-finite'): | |
| audit.prediction_parity(rows, rows) | |
| def test_recursive_state_catches_adam_rng_and_dtype_changes(self): | |
| state = {'optimizer': {0: {'exp_avg': torch.tensor([1.0]), 'step': torch.tensor(8.0)}}, | |
| 'rng': [torch.tensor([1, 2], dtype=torch.uint8)], 'python': (3, (1, 2), None)} | |
| audit.exact_state(state, copy.deepcopy(state), torch) | |
| for key in ('optimizer', 'rng', 'dtype'): | |
| changed = copy.deepcopy(state) | |
| if key == 'optimizer': changed[key][0]['exp_avg'][0] += 1 | |
| elif key == 'rng': changed[key][0][0] += 1 | |
| else: changed['rng'][0] = changed['rng'][0].long() | |
| with self.subTest(key=key), self.assertRaises(ValueError): | |
| audit.exact_state(state, changed, torch) | |
| def test_adam_requires_all_parameter_states_at_pilot_step(self): | |
| artifact = {'optimizer': {'param_groups': [{'params': [0, 1]}], 'state': { | |
| i: {'step': torch.tensor(8.0), 'exp_avg': torch.zeros(2), 'exp_avg_sq': torch.ones(2)} | |
| for i in (0, 1)}}} | |
| self.assertEqual(audit.optimizer_steps(artifact, 8, torch)['parameter_states'], 2) | |
| for kind in ('old_step', 'missing', 'nonfinite'): | |
| changed = copy.deepcopy(artifact) | |
| if kind == 'old_step': changed['optimizer']['state'][0]['step'] = 1508 | |
| elif kind == 'missing': del changed['optimizer']['state'][1] | |
| else: changed['optimizer']['state'][1]['exp_avg'][0] = float('inf') | |
| with self.subTest(kind=kind), self.assertRaises(ValueError): | |
| audit.optimizer_steps(changed, 8, torch) | |
| def test_finite_updates_need_contiguous_post_resume_rows_and_cap(self): | |
| with tempfile.TemporaryDirectory() as directory: | |
| path = Path(directory)/'training.jsonl' | |
| rows = [{'step': n, 'loss': 0.2, 'gradient_norm': 0.1, | |
| 'peak_cuda_allocated_bytes': audit.CAP} for n in range(9, 13)] | |
| def write(values): path.write_text(''.join(json.dumps(r)+'\n' for r in values)) | |
| write(rows) | |
| self.assertEqual(audit.finite_updates(path, 8, 4)['last_step'], 12) | |
| with path.open('a') as f: f.write('{"step":') | |
| self.assertEqual(audit.finite_updates(path, 8, 4)['count'], 4) | |
| for kind in ('nan', 'cap', 'gap', 'insufficient'): | |
| changed = copy.deepcopy(rows) | |
| if kind == 'nan': changed[0]['gradient_norm'] = float('nan') | |
| elif kind == 'cap': changed[0]['peak_cuda_allocated_bytes'] += 1 | |
| elif kind == 'gap': changed[0]['step'] = 7 | |
| else: changed.pop() | |
| write(changed) | |
| with self.subTest(kind=kind), self.assertRaises(ValueError): | |
| audit.finite_updates(path, 8, 4) | |
| def test_checkpoint_reader_is_cpu_only_and_rejects_nonfinite_weights(self): | |
| with tempfile.TemporaryDirectory() as directory: | |
| path = Path(directory)/'checkpoint.pt' | |
| artifact = {'format': 'opensysone-adapter-v1', 'trainable_state': {'head': torch.ones(2)}} | |
| torch.save(artifact, path) | |
| loaded, digest = audit.load_checkpoint(path, torch) | |
| self.assertEqual(loaded['trainable_state']['head'].device.type, 'cpu') | |
| self.assertEqual(digest, audit.sha256(path)) | |
| artifact['trainable_state']['head'][0] = float('nan') | |
| torch.save(artifact, path) | |
| with self.assertRaisesRegex(ValueError, 'tensor'): | |
| audit.load_checkpoint(path, torch) | |
| self.assertFalse(torch.cuda.is_initialized()) | |
| if __name__ == '__main__': | |
| unittest.main() | |