timteh673's picture
Publish release card, license, benchmarks, and integrity manifests
2e762e9 verified
Raw History Blame Contribute Delete
6.09 kB
{
"code_termination_diagnostic": {
"among_outputs_reaching_execution": {
"control": {
"executed": 103,
"pass_percent": 15.53,
"passed": 16
},
"winner": {
"executed": 46,
"pass_percent": 21.74,
"passed": 10
}
},
"interpretation": "The full-code result is dominated by severe termination/extraction pathology and is not a clean latent-code estimate.",
"winner_generations_hitting_512_token_cap": 376,
"winner_generations_total": 421
},
"comparison": {
"benign_kl": {
"control": 0.0,
"strict_limit": 0.05,
"winner": 0.093614
},
"capability_macro_percent": {
"control": 17.6859,
"winner": 21.0086
},
"exact_prompt_echoes": {
"control": 2,
"leakage_detected": true,
"winner": 3
},
"full_code_passes": {
"control": 16,
"denominator": 421,
"winner": 10
},
"harmful_hard_refusal_percent": {
"control": 43.2,
"winner": 0.0
},
"harmful_soft_deflection_percent": {
"control": 14.6,
"winner": 0.2
},
"harmful_substantive_response_percent": {
"control": 47.0,
"winner": 99.4
},
"held_out_loss_ratio": {
"control": 1.0,
"limit": 1.05,
"winner": 1.024478
},
"human_eval_percent": {
"control": 7.9268,
"winner": 4.2683
},
"incoherence_percent": {
"control": 2.7692,
"strict_winner_limit": 2.7692,
"winner": 4.3077
},
"long_form_pass_percent": {
"control": 54.1667,
"winner": 62.5
},
"max_repeated_4gram_fraction_percent": {
"strict_limit": 5.0,
"winner": 5.8632
},
"mmmu30_correct": {
"control": 9,
"denominator": 30,
"winner": 11
}
},
"evaluation_provenance": {
"kind": "self-run_frozen_local_benchmarks",
"note": "Results were produced by this project on a frozen local evaluation suite. They must not be mixed with or represented as official Qwen benchmark results.",
"official_qwen_benchmarks": false,
"release_report_frozen_at": "2026-08-22T00:01:48.742409+00:00"
},
"lineage": {
"abliterix_version": "1.12.2",
"base_model": "Qwen/Qwen3.8-27B",
"method": "Reasoning QLoRA merge followed by one selected Abliterix pass for the winner; control received no Abliterix residual-writer edits.",
"reasoning_control": "control-bf16",
"seed": 42,
"winner": "abliterix-pass1-bf16"
},
"native_mlx_proof": {
"control_mtp_statistics": {
"accepted_drafts_per_round": 1.73,
"accepted_tokens_per_round": 2.73,
"average_draft": 2.0,
"drafted_percent": 86.5,
"rounds": 37
},
"mtp_exact_answer": {
"control": "323",
"winner": "323"
},
"ordinary_exact_answer": {
"control": "323",
"winner": "323"
},
"runtime": "mlx-vlm on arm64 macOS/Metal",
"speed_note": "MTP speed is workload-dependent; positive acceptance and exact-answer agreement are the correctness evidence.",
"winner_mtp_statistics": {
"accepted_drafts_per_round": 1.78,
"accepted_tokens_per_round": 2.78,
"average_draft": 2.0,
"drafted_percent": 88.8,
"rounds": 76
},
"winner_quantization": {
"bits": 8,
"group_size": 64,
"scheme": "affine"
}
},
"release_family": "Qwen3.8-27B Opus Personal Model v1",
"schema_version": 1,
"selection": {
"all_preregistered_strict_gates_passed": false,
"control_variant": "control-bf16",
"selected_variant": "abliterix-pass1-bf16",
"status": "selected_practical_winner_with_measured_deviations",
"summary": "Practical personal-model selection with measured deviations; not universal dominance."
},
"strict_deviations": [
{
"metric": "benign_kl",
"passed": false,
"strict_limit": 0.05,
"winner": 0.093614
},
{
"metric": "incoherence_percent",
"passed": false,
"strict_limit": 2.7692,
"winner": 4.3077
},
{
"control": 7.9268,
"metric": "human_eval_percent",
"passed": false,
"strict_delta_floor": -3.0,
"winner": 4.2683,
"winner_delta_points": -3.6585
},
{
"control": "16/421",
"metric": "full_code_passes",
"passed": false,
"winner": "10/421"
},
{
"metric": "max_repeated_4gram_fraction_percent",
"passed": false,
"strict_limit": 5.0,
"winner": 5.8632
},
{
"control_exact_echoes": 2,
"metric": "prompt_leakage",
"passed": false,
"winner_exact_echoes": 3
}
],
"tensor_integrity": {
"native_mtp_tensors": 15,
"total_tensor_keys": 1199,
"unexpected_changes": 0,
"vision_tensors_preserved": 333,
"winner_abliterix_residual_writer_edits": 74
},
"training": {
"dataset_preparation": {
"accepted_rows": 12614,
"duplicates_removed": 208,
"invalid_rows_removed": 20,
"processed_manifest_sha256": "6e0a36ad20732c5f98ff592c4565a4c86876fead4e9f94bc6ceedfad1339a94d",
"raw_rows": 12842,
"rows_published": false,
"source_rows": {
"high-reasoning-250x": 250,
"opus-10000x": 9633,
"opus-3000x": 2326,
"reasoning-700x": 633
},
"split_sha256": {
"test.jsonl": "a2b6b46ce8b054166b0e392701e49390a44e391b56d25a0a25e9b8a9000257d9",
"train.jsonl": "c7e691bcabd945e1259537f74119bce4c2c8f5fa4d2fba2e626b95bc04176ad3",
"validation.jsonl": "589a05d0cf2fededff1febb065d6b3c8c947d29e948495c1567bcd9f5c44f364"
},
"splits": {
"test": 138,
"train": 12349,
"validation": 127
}
},
"dtype_after_merge": "bfloat16",
"final_validation_loss": 0.23739749,
"hardware": "NVIDIA H200-class run; the sealed Abliterix environment recorded NVIDIA H200, CUDA 13.0, driver 580.159.03, Python 3.11.15.",
"optimizer_steps": 1544,
"rows": 12349,
"token_accuracy_percent": 91.7594,
"trainable_lora_parameters": 108789760
}
}