File size: 8,546 Bytes
0dcf39e | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 | """Validation์์ exact/family logit ๊ฒฐํฉ ๊ฐ์ค์น๋ฅผ ๊ณ ์ ํ๊ณ paired-test์ ํ ๋ฒ ์ ์ฉํ๋ค."""
from __future__ import annotations
import argparse
from datetime import datetime, timezone
import json
import math
from pathlib import Path
import sys
from typing import Sequence
import torch
PROJECT_ROOT = Path(__file__).parents[1]
SOURCE_ROOT = PROJECT_ROOT / "src"
for path in (PROJECT_ROOT, SOURCE_ROOT):
if str(path) not in sys.path:
sys.path.insert(0, str(path))
from scripts.train_math_ink_06_p_boundary_auxiliary import _load_encoder06
from scripts.train_math_ink_06_skeleton_adapter import _resolve_device06
from math_grid_drawer.research.trajectory_sequence import shape_family
def _parse_args() -> argparse.Namespace:
"""ํ์ ๋ณ์: validation/test cache์ ์ธ seed. ์๋ ์๋ฆฌ: test ์ ํ์ ๊ธ์งํ calibration CLI๋ฅผ ๋ง๋ ๋ค."""
parser = argparse.ArgumentParser(description="Calibrate Math Ink 0.6 online family fusion")
parser.add_argument("--validation-cache", type=Path, required=True)
parser.add_argument("--test-cache", type=Path, required=True)
parser.add_argument("--base-checkpoint", type=Path, action="append", required=True)
parser.add_argument("--adapter-checkpoint", type=Path, action="append", required=True)
parser.add_argument("--weights", default="0,0.025,0.05,0.075,0.1,0.125,0.15,0.2,0.25,0.3")
parser.add_argument("--batch-size", type=int, default=256)
parser.add_argument("--device", choices=("auto", "cpu", "cuda"), default="auto")
parser.add_argument("--output", type=Path, required=True)
args = parser.parse_args()
if len(args.base_checkpoint) != len(args.adapter_checkpoint):
raise ValueError("base์ adapter checkpoint ๊ฐ์๋ ๊ฐ์์ผ ํฉ๋๋ค.")
if len(args.base_checkpoint) < 2:
raise ValueError("fusion calibration์๋ seed ๋ ๊ฐ ์ด์์ด ํ์ํฉ๋๋ค.")
return args
def _macro_f106(targets: torch.Tensor, predictions: torch.Tensor) -> float:
"""ํ์ ๋ณ์: ์ ๋ตยท์์ธก index. ์๋ ์๋ฆฌ: test์ ์๋ class๋ฅผ ๋ถ๋ชจ์์ ์ ์ธํ macro-F1์ ๊ณ์ฐํ๋ค."""
values = []
for label in targets.unique().tolist():
truth = targets.eq(label)
predicted = predictions.eq(label)
true_positive = int((truth & predicted).sum())
denominator = 2 * true_positive + int((truth & ~predicted).sum()) + int((~truth & predicted).sum())
values.append(2 * true_positive / denominator if denominator else 0.0)
return sum(values) / max(len(values), 1)
def _metrics06(logits: torch.Tensor, targets: torch.Tensor) -> dict[str, float | int]:
"""ํ์ ๋ณ์: fused logitยท์ ๋ต. ์๋ ์๋ฆฌ: ๋์ผ ๋ถ๋ชจ์ top-1/top-5/macro-F1์ ๋ฐํํ๋ค."""
prediction = logits.argmax(dim=-1)
top5 = logits.topk(min(5, logits.shape[-1]), dim=-1).indices
return {
"samples": len(targets),
"top1": float(prediction.eq(targets).float().mean()),
"top5": float(top5.eq(targets[:, None]).any(dim=-1).float().mean()),
"macro_f1": _macro_f106(targets, prediction),
}
def fusion_sweep06(
exact_by_seed: Sequence[torch.Tensor],
family_by_seed: Sequence[torch.Tensor],
targets: torch.Tensor,
exact_family_index: torch.Tensor,
weights: Sequence[float],
) -> list[dict[str, float | int]]:
"""ํ์ ๋ณ์: seed๋ณ exact/family logitยท๊ฐ์ค์น. ์๋ ์๋ฆฌ: ํ๋ฅ ๊ณต๊ฐ seed ensemble์ weight๋ณ ํ๊ฐํ๋ค."""
if len(exact_by_seed) != len(family_by_seed) or not exact_by_seed:
raise ValueError("exact/family seed ์ถ๋ ฅ ๊ฐ์๊ฐ ์ฌ๋ฐ๋ฅด์ง ์์ต๋๋ค.")
rows = []
for weight in weights:
seed_joint = []
for exact, family in zip(exact_by_seed, family_by_seed, strict=True):
joint = exact.log_softmax(dim=-1)
if weight:
joint = joint + float(weight) * family.log_softmax(dim=-1)[:, exact_family_index]
seed_joint.append(joint)
ensemble = torch.logsumexp(torch.stack(seed_joint), dim=0) - math.log(len(seed_joint))
rows.append({"family_fusion_weight": float(weight), **_metrics06(ensemble, targets)})
return rows
def _infer_split06(
cache_path: Path,
base_paths: Sequence[Path],
adapter_paths: Sequence[Path],
*,
device: torch.device,
batch_size: int,
) -> tuple[list[torch.Tensor], list[torch.Tensor], torch.Tensor, torch.Tensor]:
"""ํ์ ๋ณ์: split cacheยทcomposite seed. ์๋ ์๋ฆฌ: fusion ์ exact/family logit๊ณผ ์ฌ์์ ์์งํ๋ค."""
cache = torch.load(cache_path, map_location="cpu", weights_only=True, mmap=True)
features = cache["features"][:, 0]
targets = cache["targets"].long().clone()
exact_rows, family_rows = [], []
family_index: torch.Tensor | None = None
for base_path, adapter_path in zip(base_paths, adapter_paths, strict=True):
model, adapter, base, _adapter_payload = _load_encoder06(base_path, adapter_path, device)
family_to_index = {
str(label): index for index, label in enumerate(base["family_labels"])
}
current_family_index = torch.tensor([
family_to_index[shape_family(str(label))]
for label in base["exact_labels"]
], dtype=torch.long)
if family_index is not None and not torch.equal(family_index, current_family_index):
raise ValueError("seed๋ณ exactโfamily ์ฌ์์ด ๋ค๋ฆ
๋๋ค.")
family_index = current_family_index
exact_batches, family_batches = [], []
model.eval()
adapter.eval()
with torch.inference_mode():
for start in range(0, len(features), batch_size):
exact, family = model.forward_online(
adapter(features[start:start + batch_size].to(device)),
)
exact_batches.append(exact.cpu())
family_batches.append(family.cpu())
exact_rows.append(torch.cat(exact_batches))
family_rows.append(torch.cat(family_batches))
del model, adapter
if device.type == "cuda":
torch.cuda.empty_cache()
assert family_index is not None
return exact_rows, family_rows, targets, family_index
def main() -> None:
"""ํ์ ๋ณ์: CLI ์ค์ . ์๋ ์๋ฆฌ: validation winner๋ง test์ ์ ์ฉํ๊ณ ๊ฒฐ๊ณผ๋ฅผ UTF-8 JSON์ผ๋ก ๊ณ ์ ํ๋ค."""
args = _parse_args()
weights = tuple(float(value.strip()) for value in args.weights.split(",") if value.strip())
if not weights or any(weight < 0.0 or weight > 1.0 for weight in weights):
raise ValueError("family fusion weight๋ 0~1 ๋ฒ์์ฌ์ผ ํฉ๋๋ค.")
device = _resolve_device06(args.device)
validation = _infer_split06(
args.validation_cache, args.base_checkpoint, args.adapter_checkpoint,
device=device, batch_size=args.batch_size,
)
validation_sweep = fusion_sweep06(*validation, weights)
selected = max(
validation_sweep,
key=lambda row: (float(row["top1"]), float(row["macro_f1"]), -float(row["family_fusion_weight"])),
)
test = _infer_split06(
args.test_cache, args.base_checkpoint, args.adapter_checkpoint,
device=device, batch_size=args.batch_size,
)
test_result = fusion_sweep06(
*test, (float(selected["family_fusion_weight"]),),
)[0]
zero_test = fusion_sweep06(*test, (0.0,))[0]
report = {
"experiment": "MATH-INK-06-ONLINE-FAMILY-FUSION-CALIBRATION-001",
"generated_at": datetime.now(timezone.utc).isoformat(),
"selection_contract": "validation only; paired-test evaluated once after weight lock",
"device": str(device),
"validation_sweep": validation_sweep,
"selected_validation": selected,
"paired_test_zero_weight": zero_test,
"paired_test_selected_weight": test_result,
"paired_test_gain_pp": {
"top1": (float(test_result["top1"]) - float(zero_test["top1"])) * 100.0,
"top5": (float(test_result["top5"]) - float(zero_test["top5"])) * 100.0,
"macro_f1": (float(test_result["macro_f1"]) - float(zero_test["macro_f1"])) * 100.0,
},
"product_validation": False,
}
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(
json.dumps(report, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
print(json.dumps(report, ensure_ascii=False, indent=2))
if __name__ == "__main__":
main()
|