{ "gui_run": "/home/andy/ai/opensysone/runs/20260917T032753Z-playground", "gui_pid": 1467162, "gui_source_commit": "e94e58ee311ab10ebe26fc46d4a5dde2280679f6", "trainer_campaign": "/home/andy/ai/opensysone/runs/20260916T193741Z-24h", "purpose": "Read-only coexistence observation after browser tests switched three real frozen models", "observation_method": "Two process/memory/status samples about45seconds apart; GET /api/status only; no scoring/model requests from observer", "launch_preflight_mem_available_bytes": 106415841280, "samples": [ { "observed_utc": "2026-09-17T03:35:55.926185+00:00", "free_b": { "exit_code": 0, "stdout": "total used free shared buff/cache available\nMem: 130594160640 43303702528 29233373184 164196352 36601188352 87290458112\nSwap: 0 0 0" }, "gpu_processes": { "exit_code": 0, "stdout": "4674, /home/andy/ai/apps/llama.cpp/build/bin/llama-server, 170 MiB\n1096505, /home/andy/ai/envs/opensysone/bin/python, 16315 MiB\n1467162, /home/andy/ai/envs/opensysone/bin/python, 15655 MiB" }, "gpu_status": { "exit_code": 0, "stdout": "71, 93 %, 44.89 W" }, "processes": [ { "role": "gui", "pid": 1467162, "command_matches": true, "oom_score_adj": "0", "status": { "State": "S (sleeping)", "VmHWM": "32281892 kB", "VmRSS": "24261080 kB", "Threads": "56" } }, { "role": "trainer", "pid": 1096505, "command_matches": true, "oom_score_adj": "0", "status": { "State": "S (sleeping)", "VmHWM": "24653352 kB", "VmRSS": "3402476 kB", "Threads": "57" } } ], "gui_status": { "http_status": 200, "body": { "status": "ready", "loaded_model_id": "gx10-4b" } }, "gui_error_marker_counts": { "model_load_failed": 0, "model_score_failed": 0, "Traceback": 0, "OutOfMemoryError": 0 }, "training": { "last_step": 3063, "last_update": { "step": 3063, "loss": 0.014626714604673907, "gradient_norm": 0.30683499574661255, "seconds": 8.4302611759922, "decisions": 4, "branches": 12, "actual_branch_tokens": 1218, "padded_branch_tokens": 1218, "peak_cuda_allocated_bytes": 16776178176, "lr_factor": 0.977023980559572 }, "recent100_finite": true, "recent100_median_step_seconds": 9.043510261006304, "recent100_mean_step_seconds": 9.17212467954203, "recent100_max_step_seconds": 11.7679517439974, "recent4_updates": [ { "step": 3060, "loss": 0.019744501798413694, "gradient_norm": 0.9511153101921082, "seconds": 11.7679517439974, "decisions": 4, "branches": 13, "actual_branch_tokens": 1523, "padded_branch_tokens": 1523, "peak_cuda_allocated_bytes": 16776178176, "lr_factor": 0.9770701241265155 }, { "step": 3061, "loss": 0.3342024694720731, "gradient_norm": 11.301324844360352, "seconds": 11.187965510995127, "decisions": 4, "branches": 11, "actual_branch_tokens": 1641, "padded_branch_tokens": 1641, "peak_cuda_allocated_bytes": 16776178176, "lr_factor": 0.977054747970101 }, { "step": 3062, "loss": 0.02495433943113312, "gradient_norm": 0.9082462191581726, "seconds": 9.814839319995372, "decisions": 4, "branches": 14, "actual_branch_tokens": 1375, "padded_branch_tokens": 1375, "peak_cuda_allocated_bytes": 16776178176, "lr_factor": 0.9770393667810657 }, { "step": 3063, "loss": 0.014626714604673907, "gradient_norm": 0.30683499574661255, "seconds": 8.4302611759922, "decisions": 4, "branches": 12, "actual_branch_tokens": 1218, "padded_branch_tokens": 1218, "peak_cuda_allocated_bytes": 16776178176, "lr_factor": 0.977023980559572 } ], "peak_cuda_allocated_bytes": 16776178176, "checkpoint_age_seconds": 585.4080212116241, "heartbeat_utc": "2026-09-17T03:35:52.527546+00:00" } }, { "observed_utc": "2026-09-17T03:36:41.021859+00:00", "free_b": { "exit_code": 0, "stdout": "total used free shared buff/cache available\nMem: 130594160640 43292459008 29244313600 164196352 36601479168 87301701632\nSwap: 0 0 0" }, "gpu_processes": { "exit_code": 0, "stdout": "4674, /home/andy/ai/apps/llama.cpp/build/bin/llama-server, 170 MiB\n1096505, /home/andy/ai/envs/opensysone/bin/python, 16315 MiB\n1467162, /home/andy/ai/envs/opensysone/bin/python, 15655 MiB" }, "gpu_status": { "exit_code": 0, "stdout": "71, 93 %, 45.40 W" }, "processes": [ { "role": "gui", "pid": 1467162, "command_matches": true, "oom_score_adj": "0", "status": { "State": "S (sleeping)", "VmHWM": "32281892 kB", "VmRSS": "24261080 kB", "Threads": "56" } }, { "role": "trainer", "pid": 1096505, "command_matches": true, "oom_score_adj": "0", "status": { "State": "S (sleeping)", "VmHWM": "24653352 kB", "VmRSS": "3402596 kB", "Threads": "57" } } ], "gui_status": { "http_status": 200, "body": { "status": "ready", "loaded_model_id": "gx10-4b" } }, "gui_error_marker_counts": { "model_load_failed": 0, "model_score_failed": 0, "Traceback": 0, "OutOfMemoryError": 0 }, "training": { "last_step": 3068, "last_update": { "step": 3068, "loss": 0.001936116958859202, "gradient_norm": 0.10404492169618607, "seconds": 8.810512408002978, "decisions": 4, "branches": 12, "actual_branch_tokens": 1407, "padded_branch_tokens": 1407, "peak_cuda_allocated_bytes": 16776178176, "lr_factor": 0.9769469739709091 }, "recent100_finite": true, "recent100_median_step_seconds": 9.03708774750703, "recent100_mean_step_seconds": 9.169983857631742, "recent100_max_step_seconds": 11.7679517439974, "recent4_updates": [ { "step": 3065, "loss": 0.15298267499747453, "gradient_norm": 6.990951061248779, "seconds": 8.907861668005353, "decisions": 4, "branches": 10, "actual_branch_tokens": 1726, "padded_branch_tokens": 1726, "peak_cuda_allocated_bytes": 16776178176, "lr_factor": 0.9769931930198583 }, { "step": 3066, "loss": 0.0036661182439274853, "gradient_norm": 0.1209888607263565, "seconds": 8.316543419990921, "decisions": 4, "branches": 11, "actual_branch_tokens": 1353, "padded_branch_tokens": 1353, "peak_cuda_allocated_bytes": 16776178176, "lr_factor": 0.9769777917019633 }, { "step": 3067, "loss": 0.1977057716858326, "gradient_norm": 11.540794372558594, "seconds": 9.568666854989715, "decisions": 4, "branches": 11, "actual_branch_tokens": 1853, "padded_branch_tokens": 1853, "peak_cuda_allocated_bytes": 16776178176, "lr_factor": 0.9769623853522593 }, { "step": 3068, "loss": 0.001936116958859202, "gradient_norm": 0.10404492169618607, "seconds": 8.810512408002978, "decisions": 4, "branches": 12, "actual_branch_tokens": 1407, "padded_branch_tokens": 1407, "peak_cuda_allocated_bytes": 16776178176, "lr_factor": 0.9769469739709091 } ], "peak_cuda_allocated_bytes": 16776178176, "checkpoint_age_seconds": 630.5036952495575, "heartbeat_utc": "2026-09-17T03:36:37.538019+00:00" } } ], "observation_window": { "elapsed_seconds": 45.095674, "optimizer_updates_completed": 5, "steps_per_minute": 6.6525228118333475, "new_update_rows": [ { "step": 3064, "loss": 0.008182858524378389, "gradient_norm": 0.19264547526836395, "seconds": 9.048552577005466, "decisions": 4, "branches": 13, "actual_branch_tokens": 1311, "padded_branch_tokens": 1311, "peak_cuda_allocated_bytes": 16776178176, "lr_factor": 0.977008589305782 }, { "step": 3065, "loss": 0.15298267499747453, "gradient_norm": 6.990951061248779, "seconds": 8.907861668005353, "decisions": 4, "branches": 10, "actual_branch_tokens": 1726, "padded_branch_tokens": 1726, "peak_cuda_allocated_bytes": 16776178176, "lr_factor": 0.9769931930198583 }, { "step": 3066, "loss": 0.0036661182439274853, "gradient_norm": 0.1209888607263565, "seconds": 8.316543419990921, "decisions": 4, "branches": 11, "actual_branch_tokens": 1353, "padded_branch_tokens": 1353, "peak_cuda_allocated_bytes": 16776178176, "lr_factor": 0.9769777917019633 }, { "step": 3067, "loss": 0.1977057716858326, "gradient_norm": 11.540794372558594, "seconds": 9.568666854989715, "decisions": 4, "branches": 11, "actual_branch_tokens": 1853, "padded_branch_tokens": 1853, "peak_cuda_allocated_bytes": 16776178176, "lr_factor": 0.9769623853522593 }, { "step": 3068, "loss": 0.001936116958859202, "gradient_norm": 0.10404492169618607, "seconds": 8.810512408002978, "decisions": 4, "branches": 12, "actual_branch_tokens": 1407, "padded_branch_tokens": 1407, "peak_cuda_allocated_bytes": 16776178176, "lr_factor": 0.9769469739709091 } ], "new_loss_and_gradients_all_finite": true, "mean_update_seconds": 8.930427385598886, "median_update_seconds": 8.907861668005353, "max_update_seconds": 9.568666854989715 }, "limitations": [ "No controlled baseline; recent100-step durations may include GUI activity.", "GPU process memory from nvidia-smi includes context/cache overhead and is not the PyTorch allocated-memory cap.", "Short observation verifies continued finite training and measured coexistence, not zero slowdown or long-term safety." ], "observer_loaded_models": false, "observer_changed_processes": false, "observer_accessed_credentials": false, "reserved_predictions_accessed": false }