opensysone / source /results /20260917-playground /concurrent-training.json
andyshu's picture
Back up verified OpenSysOne training snapshot and pinned source
cc3f990 verified
Raw History Blame
11.8 kB
{
"gui_run": "/home/andy/ai/opensysone/runs/20260917T032753Z-playground",
"gui_pid": 1467162,
"gui_source_commit": "e94e58ee311ab10ebe26fc46d4a5dde2280679f6",
"trainer_campaign": "/home/andy/ai/opensysone/runs/20260916T193741Z-24h",
"purpose": "Read-only coexistence observation after browser tests switched three real frozen models",
"observation_method": "Two process/memory/status samples about45seconds apart; GET /api/status only; no scoring/model requests from observer",
"launch_preflight_mem_available_bytes": 106415841280,
"samples": [
{
"observed_utc": "2026-09-17T03:35:55.926185+00:00",
"free_b": {
"exit_code": 0,
"stdout": "total used free shared buff/cache available\nMem: 130594160640 43303702528 29233373184 164196352 36601188352 87290458112\nSwap: 0 0 0"
},
"gpu_processes": {
"exit_code": 0,
"stdout": "4674, /home/andy/ai/apps/llama.cpp/build/bin/llama-server, 170 MiB\n1096505, /home/andy/ai/envs/opensysone/bin/python, 16315 MiB\n1467162, /home/andy/ai/envs/opensysone/bin/python, 15655 MiB"
},
"gpu_status": {
"exit_code": 0,
"stdout": "71, 93 %, 44.89 W"
},
"processes": [
{
"role": "gui",
"pid": 1467162,
"command_matches": true,
"oom_score_adj": "0",
"status": {
"State": "S (sleeping)",
"VmHWM": "32281892 kB",
"VmRSS": "24261080 kB",
"Threads": "56"
}
},
{
"role": "trainer",
"pid": 1096505,
"command_matches": true,
"oom_score_adj": "0",
"status": {
"State": "S (sleeping)",
"VmHWM": "24653352 kB",
"VmRSS": "3402476 kB",
"Threads": "57"
}
}
],
"gui_status": {
"http_status": 200,
"body": {
"status": "ready",
"loaded_model_id": "gx10-4b"
}
},
"gui_error_marker_counts": {
"model_load_failed": 0,
"model_score_failed": 0,
"Traceback": 0,
"OutOfMemoryError": 0
},
"training": {
"last_step": 3063,
"last_update": {
"step": 3063,
"loss": 0.014626714604673907,
"gradient_norm": 0.30683499574661255,
"seconds": 8.4302611759922,
"decisions": 4,
"branches": 12,
"actual_branch_tokens": 1218,
"padded_branch_tokens": 1218,
"peak_cuda_allocated_bytes": 16776178176,
"lr_factor": 0.977023980559572
},
"recent100_finite": true,
"recent100_median_step_seconds": 9.043510261006304,
"recent100_mean_step_seconds": 9.17212467954203,
"recent100_max_step_seconds": 11.7679517439974,
"recent4_updates": [
{
"step": 3060,
"loss": 0.019744501798413694,
"gradient_norm": 0.9511153101921082,
"seconds": 11.7679517439974,
"decisions": 4,
"branches": 13,
"actual_branch_tokens": 1523,
"padded_branch_tokens": 1523,
"peak_cuda_allocated_bytes": 16776178176,
"lr_factor": 0.9770701241265155
},
{
"step": 3061,
"loss": 0.3342024694720731,
"gradient_norm": 11.301324844360352,
"seconds": 11.187965510995127,
"decisions": 4,
"branches": 11,
"actual_branch_tokens": 1641,
"padded_branch_tokens": 1641,
"peak_cuda_allocated_bytes": 16776178176,
"lr_factor": 0.977054747970101
},
{
"step": 3062,
"loss": 0.02495433943113312,
"gradient_norm": 0.9082462191581726,
"seconds": 9.814839319995372,
"decisions": 4,
"branches": 14,
"actual_branch_tokens": 1375,
"padded_branch_tokens": 1375,
"peak_cuda_allocated_bytes": 16776178176,
"lr_factor": 0.9770393667810657
},
{
"step": 3063,
"loss": 0.014626714604673907,
"gradient_norm": 0.30683499574661255,
"seconds": 8.4302611759922,
"decisions": 4,
"branches": 12,
"actual_branch_tokens": 1218,
"padded_branch_tokens": 1218,
"peak_cuda_allocated_bytes": 16776178176,
"lr_factor": 0.977023980559572
}
],
"peak_cuda_allocated_bytes": 16776178176,
"checkpoint_age_seconds": 585.4080212116241,
"heartbeat_utc": "2026-09-17T03:35:52.527546+00:00"
}
},
{
"observed_utc": "2026-09-17T03:36:41.021859+00:00",
"free_b": {
"exit_code": 0,
"stdout": "total used free shared buff/cache available\nMem: 130594160640 43292459008 29244313600 164196352 36601479168 87301701632\nSwap: 0 0 0"
},
"gpu_processes": {
"exit_code": 0,
"stdout": "4674, /home/andy/ai/apps/llama.cpp/build/bin/llama-server, 170 MiB\n1096505, /home/andy/ai/envs/opensysone/bin/python, 16315 MiB\n1467162, /home/andy/ai/envs/opensysone/bin/python, 15655 MiB"
},
"gpu_status": {
"exit_code": 0,
"stdout": "71, 93 %, 45.40 W"
},
"processes": [
{
"role": "gui",
"pid": 1467162,
"command_matches": true,
"oom_score_adj": "0",
"status": {
"State": "S (sleeping)",
"VmHWM": "32281892 kB",
"VmRSS": "24261080 kB",
"Threads": "56"
}
},
{
"role": "trainer",
"pid": 1096505,
"command_matches": true,
"oom_score_adj": "0",
"status": {
"State": "S (sleeping)",
"VmHWM": "24653352 kB",
"VmRSS": "3402596 kB",
"Threads": "57"
}
}
],
"gui_status": {
"http_status": 200,
"body": {
"status": "ready",
"loaded_model_id": "gx10-4b"
}
},
"gui_error_marker_counts": {
"model_load_failed": 0,
"model_score_failed": 0,
"Traceback": 0,
"OutOfMemoryError": 0
},
"training": {
"last_step": 3068,
"last_update": {
"step": 3068,
"loss": 0.001936116958859202,
"gradient_norm": 0.10404492169618607,
"seconds": 8.810512408002978,
"decisions": 4,
"branches": 12,
"actual_branch_tokens": 1407,
"padded_branch_tokens": 1407,
"peak_cuda_allocated_bytes": 16776178176,
"lr_factor": 0.9769469739709091
},
"recent100_finite": true,
"recent100_median_step_seconds": 9.03708774750703,
"recent100_mean_step_seconds": 9.169983857631742,
"recent100_max_step_seconds": 11.7679517439974,
"recent4_updates": [
{
"step": 3065,
"loss": 0.15298267499747453,
"gradient_norm": 6.990951061248779,
"seconds": 8.907861668005353,
"decisions": 4,
"branches": 10,
"actual_branch_tokens": 1726,
"padded_branch_tokens": 1726,
"peak_cuda_allocated_bytes": 16776178176,
"lr_factor": 0.9769931930198583
},
{
"step": 3066,
"loss": 0.0036661182439274853,
"gradient_norm": 0.1209888607263565,
"seconds": 8.316543419990921,
"decisions": 4,
"branches": 11,
"actual_branch_tokens": 1353,
"padded_branch_tokens": 1353,
"peak_cuda_allocated_bytes": 16776178176,
"lr_factor": 0.9769777917019633
},
{
"step": 3067,
"loss": 0.1977057716858326,
"gradient_norm": 11.540794372558594,
"seconds": 9.568666854989715,
"decisions": 4,
"branches": 11,
"actual_branch_tokens": 1853,
"padded_branch_tokens": 1853,
"peak_cuda_allocated_bytes": 16776178176,
"lr_factor": 0.9769623853522593
},
{
"step": 3068,
"loss": 0.001936116958859202,
"gradient_norm": 0.10404492169618607,
"seconds": 8.810512408002978,
"decisions": 4,
"branches": 12,
"actual_branch_tokens": 1407,
"padded_branch_tokens": 1407,
"peak_cuda_allocated_bytes": 16776178176,
"lr_factor": 0.9769469739709091
}
],
"peak_cuda_allocated_bytes": 16776178176,
"checkpoint_age_seconds": 630.5036952495575,
"heartbeat_utc": "2026-09-17T03:36:37.538019+00:00"
}
}
],
"observation_window": {
"elapsed_seconds": 45.095674,
"optimizer_updates_completed": 5,
"steps_per_minute": 6.6525228118333475,
"new_update_rows": [
{
"step": 3064,
"loss": 0.008182858524378389,
"gradient_norm": 0.19264547526836395,
"seconds": 9.048552577005466,
"decisions": 4,
"branches": 13,
"actual_branch_tokens": 1311,
"padded_branch_tokens": 1311,
"peak_cuda_allocated_bytes": 16776178176,
"lr_factor": 0.977008589305782
},
{
"step": 3065,
"loss": 0.15298267499747453,
"gradient_norm": 6.990951061248779,
"seconds": 8.907861668005353,
"decisions": 4,
"branches": 10,
"actual_branch_tokens": 1726,
"padded_branch_tokens": 1726,
"peak_cuda_allocated_bytes": 16776178176,
"lr_factor": 0.9769931930198583
},
{
"step": 3066,
"loss": 0.0036661182439274853,
"gradient_norm": 0.1209888607263565,
"seconds": 8.316543419990921,
"decisions": 4,
"branches": 11,
"actual_branch_tokens": 1353,
"padded_branch_tokens": 1353,
"peak_cuda_allocated_bytes": 16776178176,
"lr_factor": 0.9769777917019633
},
{
"step": 3067,
"loss": 0.1977057716858326,
"gradient_norm": 11.540794372558594,
"seconds": 9.568666854989715,
"decisions": 4,
"branches": 11,
"actual_branch_tokens": 1853,
"padded_branch_tokens": 1853,
"peak_cuda_allocated_bytes": 16776178176,
"lr_factor": 0.9769623853522593
},
{
"step": 3068,
"loss": 0.001936116958859202,
"gradient_norm": 0.10404492169618607,
"seconds": 8.810512408002978,
"decisions": 4,
"branches": 12,
"actual_branch_tokens": 1407,
"padded_branch_tokens": 1407,
"peak_cuda_allocated_bytes": 16776178176,
"lr_factor": 0.9769469739709091
}
],
"new_loss_and_gradients_all_finite": true,
"mean_update_seconds": 8.930427385598886,
"median_update_seconds": 8.907861668005353,
"max_update_seconds": 9.568666854989715
},
"limitations": [
"No controlled baseline; recent100-step durations may include GUI activity.",
"GPU process memory from nvidia-smi includes context/cache overhead and is not the PyTorch allocated-memory cap.",
"Short observation verifies continued finite training and measured coexistence, not zero slowdown or long-term safety."
],
"observer_loaded_models": false,
"observer_changed_processes": false,
"observer_accessed_credentials": false,
"reserved_predictions_accessed": false
}