{ "power": "AC", "low_power_mode": false, "thermal_warning_reported": false, "other_model_service": "llama-server present, observed 0.1% CPU, approximately 15.6 GiB physical footprint", "caveat": "Desktop-session benchmark; other applications remain open. Not an isolated laboratory run.", "idle_sleep": "Temporarily inhibited for benchmark process only", "phase_offload": true, "swap_before_formal_run_gib": 19.60015869140625, "swap_note": "Swap allocated during earlier resident diagnostics; per-image swap deltas are recorded separately.", "other_model_services": "Other pre-existing Python and llama-server services left running; earlier physical footprints approximately 25.5 and 15.6 GiB.", "mid_run_observation": "At approximately 23:35 PDT both pre-existing model services were still present and around 0.1% CPU; pmset reported no thermal or performance warning. GPU contention and thermal state were not continuously instrumented.", "timing_variability": "Q8 texture was substantially faster than its preceding Q8 cases despite unchanged runtime; do not interpret the aggregate latency difference as a controlled causal estimate of quantization speed." }