Add composite CPU latency and memory proxy
Browse files- MANIFEST.json +49 -39
- MODEL_INDEX.json +8 -0
- README.md +5 -0
- reports/RESEARCH_REPORT.md +15 -1
- reports/composite_cpu_benchmark.json +92 -0
- scripts/benchmark_math_ink_06_composite.py +183 -0
MANIFEST.json
CHANGED
|
@@ -1,11 +1,11 @@
|
|
| 1 |
{
|
| 2 |
-
"schema": "aiflow-hf-research-snapshot-
|
| 3 |
-
"generated_at": "2026-07-23T20:
|
| 4 |
"track": "R_noncommercial_plus_rejected_P_proxy",
|
| 5 |
"product_validation": false,
|
| 6 |
"public_release": true,
|
| 7 |
"contains_raw_dataset": false,
|
| 8 |
-
"tests": "
|
| 9 |
"retracted_paths": [
|
| 10 |
"models/auxiliary/boundary_auxiliary_head.pt",
|
| 11 |
"models/auxiliary/seed17/boundary_joint_delta.pt",
|
|
@@ -30,19 +30,19 @@
|
|
| 30 |
},
|
| 31 |
{
|
| 32 |
"path": "reports/RESEARCH_REPORT.md",
|
| 33 |
-
"bytes":
|
| 34 |
-
"sha256": "
|
| 35 |
-
},
|
| 36 |
-
{
|
| 37 |
-
"path": "scripts/evaluate_crohme_tray_joint_selector.py",
|
| 38 |
-
"bytes": 11671,
|
| 39 |
-
"sha256": "58c175a95cdeb9fc764fb96594a76888122b8f118f2f26da1283e743b1a7cd04"
|
| 40 |
},
|
| 41 |
{
|
| 42 |
"path": "scripts/build_math_ink_06_litert_colab_bundle.py",
|
| 43 |
"bytes": 6862,
|
| 44 |
"sha256": "f008635ab59344005f65f1394f56275e4c92610cd25781a2d95243d060010718"
|
| 45 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 46 |
{
|
| 47 |
"path": "scripts/audit_math_ink_06_local_baseline_overmerge.py",
|
| 48 |
"bytes": 11900,
|
|
@@ -88,6 +88,11 @@
|
|
| 88 |
"bytes": 33727,
|
| 89 |
"sha256": "c3dfd328d9ff0a55eed2cb74657ad87e5e164968fddd7fae8c8fd92e137a4557"
|
| 90 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 91 |
{
|
| 92 |
"path": "src/math_ink_06.py",
|
| 93 |
"bytes": 58690,
|
|
@@ -215,13 +220,13 @@
|
|
| 215 |
},
|
| 216 |
{
|
| 217 |
"path": "MODEL_INDEX.json",
|
| 218 |
-
"bytes":
|
| 219 |
-
"sha256": "
|
| 220 |
},
|
| 221 |
{
|
| 222 |
"path": "README.md",
|
| 223 |
-
"bytes":
|
| 224 |
-
"sha256": "
|
| 225 |
},
|
| 226 |
{
|
| 227 |
"path": "assets/stroke_encoding.svg",
|
|
@@ -238,6 +243,16 @@
|
|
| 238 |
"bytes": 824191,
|
| 239 |
"sha256": "18aff472c1bdb25a519b1553364229c9869aed1f36a9d663cf88a1bc808fab8f"
|
| 240 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 241 |
{
|
| 242 |
"path": "reports/behavior_role_seed31.json",
|
| 243 |
"bytes": 45412,
|
|
@@ -249,34 +264,34 @@
|
|
| 249 |
"sha256": "e2072fec1b1065777f0e9218c48d27f096b22f5545e7cc8d9962939282719561"
|
| 250 |
},
|
| 251 |
{
|
| 252 |
-
"path": "reports/
|
| 253 |
-
"bytes":
|
| 254 |
-
"sha256": "
|
| 255 |
-
},
|
| 256 |
-
{
|
| 257 |
-
"path": "reports/behavior_role_seed47.json",
|
| 258 |
-
"bytes": 30091,
|
| 259 |
-
"sha256": "3d4dcbe2f2b0afb483f992dacd62aab1089bdc41d55cc259bcb7fc21b5294db0"
|
| 260 |
},
|
| 261 |
{
|
| 262 |
"path": "reports/local_baseline_guard_report.json",
|
| 263 |
"bytes": 24052,
|
| 264 |
"sha256": "8f209b2890245f01f437c26e7556e19fee76c057d449c9382e2d64a70a3764d6"
|
| 265 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 266 |
{
|
| 267 |
"path": "reports/component_competition_report.json",
|
| 268 |
"bytes": 22284,
|
| 269 |
"sha256": "4d3db5f71e5008381dd072f2c10aafb22e08f96e23f2a32f82a3a46abd9c70b9"
|
| 270 |
},
|
| 271 |
{
|
| 272 |
-
"path": "reports/
|
| 273 |
-
"bytes":
|
| 274 |
-
"sha256": "
|
| 275 |
},
|
| 276 |
{
|
| 277 |
-
"path": "models/
|
| 278 |
-
"bytes":
|
| 279 |
-
"sha256": "
|
| 280 |
},
|
| 281 |
{
|
| 282 |
"path": "models/seed31/base_378.pt",
|
|
@@ -289,14 +304,14 @@
|
|
| 289 |
"sha256": "8ea224d7bc107a4c8b6cb91c9de6598000dae463d3e6d39480c081ea728cc87f"
|
| 290 |
},
|
| 291 |
{
|
| 292 |
-
"path": "models/
|
| 293 |
-
"bytes":
|
| 294 |
-
"sha256": "
|
| 295 |
},
|
| 296 |
{
|
| 297 |
-
"path": "models/
|
| 298 |
-
"bytes":
|
| 299 |
-
"sha256": "
|
| 300 |
},
|
| 301 |
{
|
| 302 |
"path": "models/seed47/behavior_role_head.pt",
|
|
@@ -307,11 +322,6 @@
|
|
| 307 |
"path": "models/seed47/base_378.pt",
|
| 308 |
"bytes": 8638290,
|
| 309 |
"sha256": "4131b21f6b00471c1c5a82aee4a2da45e599e42dc742fc1dd0dd8afb970f8939"
|
| 310 |
-
},
|
| 311 |
-
{
|
| 312 |
-
"path": "models/seed31/online_adapter.pt",
|
| 313 |
-
"bytes": 2875654,
|
| 314 |
-
"sha256": "fedbbab64cea6755e76e0633367ee45fbcc1d5726bb9dfdb7c5fe4bbc968bdf2"
|
| 315 |
}
|
| 316 |
]
|
| 317 |
}
|
|
|
|
| 1 |
{
|
| 2 |
+
"schema": "aiflow-hf-research-snapshot-v11",
|
| 3 |
+
"generated_at": "2026-07-23T20:26:01.9922255Z",
|
| 4 |
"track": "R_noncommercial_plus_rejected_P_proxy",
|
| 5 |
"product_validation": false,
|
| 6 |
"public_release": true,
|
| 7 |
"contains_raw_dataset": false,
|
| 8 |
+
"tests": "287 passed",
|
| 9 |
"retracted_paths": [
|
| 10 |
"models/auxiliary/boundary_auxiliary_head.pt",
|
| 11 |
"models/auxiliary/seed17/boundary_joint_delta.pt",
|
|
|
|
| 30 |
},
|
| 31 |
{
|
| 32 |
"path": "reports/RESEARCH_REPORT.md",
|
| 33 |
+
"bytes": 35675,
|
| 34 |
+
"sha256": "e9064d37996ceb714d6ccea391b6cc00be05653a8b5190ee7fb3affda7f08d87"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 35 |
},
|
| 36 |
{
|
| 37 |
"path": "scripts/build_math_ink_06_litert_colab_bundle.py",
|
| 38 |
"bytes": 6862,
|
| 39 |
"sha256": "f008635ab59344005f65f1394f56275e4c92610cd25781a2d95243d060010718"
|
| 40 |
},
|
| 41 |
+
{
|
| 42 |
+
"path": "scripts/benchmark_math_ink_06_composite.py",
|
| 43 |
+
"bytes": 7618,
|
| 44 |
+
"sha256": "6c5347fe68114116e0dfd29e55e3682f47b3ce42af1412009ab7883e2076887c"
|
| 45 |
+
},
|
| 46 |
{
|
| 47 |
"path": "scripts/audit_math_ink_06_local_baseline_overmerge.py",
|
| 48 |
"bytes": 11900,
|
|
|
|
| 88 |
"bytes": 33727,
|
| 89 |
"sha256": "c3dfd328d9ff0a55eed2cb74657ad87e5e164968fddd7fae8c8fd92e137a4557"
|
| 90 |
},
|
| 91 |
+
{
|
| 92 |
+
"path": "scripts/evaluate_crohme_tray_joint_selector.py",
|
| 93 |
+
"bytes": 11671,
|
| 94 |
+
"sha256": "58c175a95cdeb9fc764fb96594a76888122b8f118f2f26da1283e743b1a7cd04"
|
| 95 |
+
},
|
| 96 |
{
|
| 97 |
"path": "src/math_ink_06.py",
|
| 98 |
"bytes": 58690,
|
|
|
|
| 220 |
},
|
| 221 |
{
|
| 222 |
"path": "MODEL_INDEX.json",
|
| 223 |
+
"bytes": 2766,
|
| 224 |
+
"sha256": "68ab1d983dc5da5cf547ba34e59ecae3023b5a66b73df7ee94be15445e428ef4"
|
| 225 |
},
|
| 226 |
{
|
| 227 |
"path": "README.md",
|
| 228 |
+
"bytes": 11920,
|
| 229 |
+
"sha256": "24c767d993cd750aa85a30708748386d7ab89c133f4dd756d3424f66c6958a14"
|
| 230 |
},
|
| 231 |
{
|
| 232 |
"path": "assets/stroke_encoding.svg",
|
|
|
|
| 243 |
"bytes": 824191,
|
| 244 |
"sha256": "18aff472c1bdb25a519b1553364229c9869aed1f36a9d663cf88a1bc808fab8f"
|
| 245 |
},
|
| 246 |
+
{
|
| 247 |
+
"path": "models/seed17/behavior_role_head.pt",
|
| 248 |
+
"bytes": 75794,
|
| 249 |
+
"sha256": "57ced82a2600bd342bba68ba0636ecc3fc98c3dc461e26a132b5264d0a3d95bd"
|
| 250 |
+
},
|
| 251 |
+
{
|
| 252 |
+
"path": "reports/behavior_role_seed47.json",
|
| 253 |
+
"bytes": 30091,
|
| 254 |
+
"sha256": "3d4dcbe2f2b0afb483f992dacd62aab1089bdc41d55cc259bcb7fc21b5294db0"
|
| 255 |
+
},
|
| 256 |
{
|
| 257 |
"path": "reports/behavior_role_seed31.json",
|
| 258 |
"bytes": 45412,
|
|
|
|
| 264 |
"sha256": "e2072fec1b1065777f0e9218c48d27f096b22f5545e7cc8d9962939282719561"
|
| 265 |
},
|
| 266 |
{
|
| 267 |
+
"path": "reports/boundary_behavior_guard_report.json",
|
| 268 |
+
"bytes": 72053,
|
| 269 |
+
"sha256": "f55a230a9f0892b9676c71db46e9e527862b3d8da776dc43db767ce7a0f6a5c7"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 270 |
},
|
| 271 |
{
|
| 272 |
"path": "reports/local_baseline_guard_report.json",
|
| 273 |
"bytes": 24052,
|
| 274 |
"sha256": "8f209b2890245f01f437c26e7556e19fee76c057d449c9382e2d64a70a3764d6"
|
| 275 |
},
|
| 276 |
+
{
|
| 277 |
+
"path": "reports/composite_cpu_benchmark.json",
|
| 278 |
+
"bytes": 2948,
|
| 279 |
+
"sha256": "1e9c75082cde0ac2c8ef303f2ff8870b3d2c772ad63027226e632b56f02ed9aa"
|
| 280 |
+
},
|
| 281 |
{
|
| 282 |
"path": "reports/component_competition_report.json",
|
| 283 |
"bytes": 22284,
|
| 284 |
"sha256": "4d3db5f71e5008381dd072f2c10aafb22e08f96e23f2a32f82a3a46abd9c70b9"
|
| 285 |
},
|
| 286 |
{
|
| 287 |
+
"path": "reports/behavior_role_3seed_summary.json",
|
| 288 |
+
"bytes": 3567,
|
| 289 |
+
"sha256": "b16049218d2d006d952f59f55642fa9f3fbfbc3cc3a730fbd7b075f507c3b251"
|
| 290 |
},
|
| 291 |
{
|
| 292 |
+
"path": "models/seed31/behavior_role_head.pt",
|
| 293 |
+
"bytes": 75794,
|
| 294 |
+
"sha256": "b7c22cf49320b90a6fd103d39cc3ff08ea7734ab83275aad66e2b529b749d9ae"
|
| 295 |
},
|
| 296 |
{
|
| 297 |
"path": "models/seed31/base_378.pt",
|
|
|
|
| 304 |
"sha256": "8ea224d7bc107a4c8b6cb91c9de6598000dae463d3e6d39480c081ea728cc87f"
|
| 305 |
},
|
| 306 |
{
|
| 307 |
+
"path": "models/seed31/online_adapter.pt",
|
| 308 |
+
"bytes": 2875654,
|
| 309 |
+
"sha256": "fedbbab64cea6755e76e0633367ee45fbcc1d5726bb9dfdb7c5fe4bbc968bdf2"
|
| 310 |
},
|
| 311 |
{
|
| 312 |
+
"path": "models/seed47/online_adapter.pt",
|
| 313 |
+
"bytes": 2876038,
|
| 314 |
+
"sha256": "46394eeebad3d27d3178f78e693e160700ba3f454d79044714b835bc51c07056"
|
| 315 |
},
|
| 316 |
{
|
| 317 |
"path": "models/seed47/behavior_role_head.pt",
|
|
|
|
| 322 |
"path": "models/seed47/base_378.pt",
|
| 323 |
"bytes": 8638290,
|
| 324 |
"sha256": "4131b21f6b00471c1c5a82aee4a2da45e599e42dc742fc1dd0dd8afb970f8939"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 325 |
}
|
| 326 |
]
|
| 327 |
}
|
MODEL_INDEX.json
CHANGED
|
@@ -68,6 +68,14 @@
|
|
| 68 |
"representative_samples": 76,
|
| 69 |
"litert_flatbuffer": false
|
| 70 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 71 |
"release_state": {
|
| 72 |
"track": "R_noncommercial_only",
|
| 73 |
"product_validation": false,
|
|
|
|
| 68 |
"representative_samples": 76,
|
| 69 |
"litert_flatbuffer": false
|
| 70 |
},
|
| 71 |
+
"composite_cpu_proxy": {
|
| 72 |
+
"report": "reports/composite_cpu_benchmark.json",
|
| 73 |
+
"online_p95_ms_range": [7.77, 10.2],
|
| 74 |
+
"raster_p95_ms_range": [19.18, 23.57],
|
| 75 |
+
"model_state_bytes": 8947108,
|
| 76 |
+
"model_state_plus_inference_growth_bytes": 26301860,
|
| 77 |
+
"android_validation": false
|
| 78 |
+
},
|
| 79 |
"release_state": {
|
| 80 |
"track": "R_noncommercial_only",
|
| 81 |
"product_validation": false,
|
README.md
CHANGED
|
@@ -207,6 +207,10 @@ Export 그래프는 이제 base-only가 아니라 online/raster modality adapter
|
|
| 207 |
|
| 208 |
`.pt2`는 Android용 `.tflite`가 아니다. LiteRT Torch 0.9.1 변환과 Android runtime parity는 아직 완료되지 않았으므로 `litert_exported=false`를 유지한다.
|
| 209 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 210 |
## 출력 범위
|
| 211 |
|
| 212 |
의도한 모바일 API:
|
|
@@ -259,6 +263,7 @@ PyTorch checkpoint와 joblib/pickle은 신뢰할 수 없는 출처에서 로드
|
|
| 259 |
- P boundary 정정 3-seed: [`reports/p_boundary_joint_sharedfix_3seed_summary.json`](reports/p_boundary_joint_sharedfix_3seed_summary.json)
|
| 260 |
- 정정 main device stress: [`reports/p_boundary_device_stress_sharedfix_3seed.json`](reports/p_boundary_device_stress_sharedfix_3seed.json)
|
| 261 |
- Seed-17 composite export: [`exports/seed17/export_manifest.json`](exports/seed17/export_manifest.json)
|
|
|
|
| 262 |
- LiteRT Colab notebook: [`colab/AIFlow_Math_Ink_06_LiteRT.ipynb`](colab/AIFlow_Math_Ink_06_LiteRT.ipynb)
|
| 263 |
- 실제 P formula schema: [`contracts/aiflow_p_formula_v1.schema.json`](contracts/aiflow_p_formula_v1.schema.json)
|
| 264 |
- 파일 checksum: [`MANIFEST.json`](MANIFEST.json)
|
|
|
|
| 207 |
|
| 208 |
`.pt2`는 Android용 `.tflite`가 아니다. LiteRT Torch 0.9.1 변환과 Android runtime parity는 아직 완료되지 않았으므로 `litert_exported=false`를 유지한다.
|
| 209 |
|
| 210 |
+
### CPU latency·memory proxy
|
| 211 |
+
|
| 212 |
+
Windows PyTorch CPU의 실제 대표 입력 76개 측정에서 online p95는 7.77~10.20ms, raster p95는 19.18~23.57ms였다. Tensor state는 8.95MB, 모델 로드 후 inference RSS 증가분을 합친 구조 proxy는 26.30MB다. 전체 Python process RSS 231.5MB는 PyTorch runtime을 포함하므로 Android LiteRT memory 근거가 아니다.
|
| 213 |
+
|
| 214 |
## 출력 범위
|
| 215 |
|
| 216 |
의도한 모바일 API:
|
|
|
|
| 263 |
- P boundary 정정 3-seed: [`reports/p_boundary_joint_sharedfix_3seed_summary.json`](reports/p_boundary_joint_sharedfix_3seed_summary.json)
|
| 264 |
- 정정 main device stress: [`reports/p_boundary_device_stress_sharedfix_3seed.json`](reports/p_boundary_device_stress_sharedfix_3seed.json)
|
| 265 |
- Seed-17 composite export: [`exports/seed17/export_manifest.json`](exports/seed17/export_manifest.json)
|
| 266 |
+
- Composite CPU benchmark: [`reports/composite_cpu_benchmark.json`](reports/composite_cpu_benchmark.json)
|
| 267 |
- LiteRT Colab notebook: [`colab/AIFlow_Math_Ink_06_LiteRT.ipynb`](colab/AIFlow_Math_Ink_06_LiteRT.ipynb)
|
| 268 |
- 실제 P formula schema: [`contracts/aiflow_p_formula_v1.schema.json`](contracts/aiflow_p_formula_v1.schema.json)
|
| 269 |
- 파일 checksum: [`MANIFEST.json`](MANIFEST.json)
|
reports/RESEARCH_REPORT.md
CHANGED
|
@@ -413,6 +413,18 @@ Linux/Colab 변환을 위해 label이 없는 실제 HWRT-derived online/raster
|
|
| 413 |
|
| 414 |
Python 공개 런타임의 `MathInk06Engine`도 base-only였으므로 `adapter_checkpoint`를 공식 생성자 인자로 추가했다. 엔진은 base→adapter `shared_state_dict`→dual modality adapter 순서로 구성하고 online은 `adapter.online`, raster virtual trajectory는 `adapter.raster`를 거친다. 실제 seed-17 representative 입력에서 엔진과 export wrapper의 online/raster 최대 logit 오차는 모두 0.0이었다. Adapter를 생략한 호출은 호환용 base-only 경로로만 남긴다.
|
| 415 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 416 |
### 실제 P 연속식 데이터 계약
|
| 417 |
|
| 418 |
실제 데이터가 도착하기 전에 `AIFlow P Formula v1` JSON Schema와 fail-closed preflight를 추가했다. 각 formula는 다음 근거를 모두 가져야 한다.
|
|
@@ -424,7 +436,7 @@ Python 공개 런타임의 `MathInk06Engine`도 base-only였으므로 `adapter_c
|
|
| 424 |
- 양수 canvas 크기
|
| 425 |
- token과 원본 raw stroke가 포함된 정답 symbol group
|
| 426 |
|
| 427 |
-
Origin·writer·device·source가 둘 이상의 split에 나타나면 제품 평가를 거부한다. Timestamp와 pressure가 없는 symbol은 삭제하거나 관측값으로 위장하지 않고 missing slice로 센다. 검증된 formula의 실제 symbol group은 boundary 음성, 인접한 두 symbol group 결합은 boundary 양성으로 만든다. 이 계약은 실제 성능값을 만들지는 않지만, 향후 P 입력이 CROHME 정답 group이나 합성 고립기호 proxy와 섞이는 것을 방지한다. 전체 회귀는
|
| 428 |
|
| 429 |
## 산출물
|
| 430 |
|
|
@@ -450,6 +462,7 @@ Origin·writer·device·source가 둘 이상의 split에 나타나면 제품 평
|
|
| 450 |
- `scripts/evaluate_math_ink_06_p_boundary_device_stress.py`
|
| 451 |
- `scripts/export_math_ink_06_litert.py`
|
| 452 |
- `scripts/build_math_ink_06_litert_colab_bundle.py`
|
|
|
|
| 453 |
- `scripts/preflight_math_ink_06_p_formula.py`
|
| 454 |
- `scripts/analyze_crohme_lattice_failures.py`
|
| 455 |
- `tests/test_behavior_context06.py`
|
|
@@ -458,6 +471,7 @@ Origin·writer·device·source가 둘 이상의 split에 나타나면 제품 평
|
|
| 458 |
- `research/contracts/aiflow_p_formula_v1.schema.json`
|
| 459 |
- `research/colab/AIFlow_Math_Ink_06_LiteRT.ipynb`
|
| 460 |
- `research/runs/math_ink_06_litert_colab_20260724/aiflow_math_ink_06_litert_bundle.zip`
|
|
|
|
| 461 |
- `research/AIFlow-MATH-INK-0.6-BEHAVIOR-CONTEXT-REPORT-20260724.md`
|
| 462 |
- `research/runs/math_ink_06_behavior_role_3seed_20260724/run_summary.json`
|
| 463 |
- `research/runs/math_ink_06_behavior_grouping_audit_20260724/report.json`
|
|
|
|
| 413 |
|
| 414 |
Python 공개 런타임의 `MathInk06Engine`도 base-only였으므로 `adapter_checkpoint`를 공식 생성자 인자로 추가했다. 엔진은 base→adapter `shared_state_dict`→dual modality adapter 순서로 구성하고 online은 `adapter.online`, raster virtual trajectory는 `adapter.raster`를 거친다. 실제 seed-17 representative 입력에서 엔진과 export wrapper의 online/raster 최대 logit 오차는 모두 0.0이었다. Adapter를 생략한 호출은 호환용 base-only 경로로만 남긴다.
|
| 415 |
|
| 416 |
+
### Composite CPU latency·memory proxy
|
| 417 |
+
|
| 418 |
+
정정된 Python runtime을 실제 representative 76개로 Windows CPU에서 측정했다. 각 thread 조건은 같은 입력과 output checksum을 사용했다.
|
| 419 |
+
|
| 420 |
+
| Intra-op threads | Online p95 | Raster p95 |
|
| 421 |
+
|---:|---:|---:|
|
| 422 |
+
| 1 | 10.20ms | 20.65ms |
|
| 423 |
+
| 2 | 8.51ms | 23.57ms |
|
| 424 |
+
| 4 | 7.77ms | 19.18ms |
|
| 425 |
+
|
| 426 |
+
세 조건 모두 online 50ms, raster 200ms software proxy를 통과했다. Model과 composite adapter의 중복 제거 tensor state는 8,947,108 bytes다. 모델 로드 후 관측한 inference RSS 증가분은 17,354,752 bytes이고 두 값을 합친 구조 proxy는 26,301,860 bytes로 100MiB 안이다. 반면 전체 Python process RSS는 최대 231,530,496 bytes였는데 이는 CPython·PyTorch runtime을 포함하므로 Android peak memory와 직접 비교하지 않는다. 실제 Android LiteRT 3-tier latency·memory·battery·delegate 검증은 계속 필요하다.
|
| 427 |
+
|
| 428 |
### 실제 P 연속식 데이터 계약
|
| 429 |
|
| 430 |
실제 데이터가 도착하기 전에 `AIFlow P Formula v1` JSON Schema와 fail-closed preflight를 추가했다. 각 formula는 다음 근거를 모두 가져야 한다.
|
|
|
|
| 436 |
- 양수 canvas 크기
|
| 437 |
- token과 원본 raw stroke가 포함된 정답 symbol group
|
| 438 |
|
| 439 |
+
Origin·writer·device·source가 둘 이상의 split에 나타나면 제품 평가를 거부한다. Timestamp와 pressure가 없는 symbol은 삭제하거나 관측값으로 위장하지 않고 missing slice로 센다. 검증된 formula의 실제 symbol group은 boundary 음성, 인접한 두 symbol group 결합은 boundary 양성으로 만든다. 이 계약은 실제 성능값을 만들지는 않지만, 향후 P 입력이 CROHME 정답 group이나 합성 고립기호 proxy와 섞이는 것을 방지한다. 전체 회귀는 287개가 통과했다.
|
| 440 |
|
| 441 |
## 산출물
|
| 442 |
|
|
|
|
| 462 |
- `scripts/evaluate_math_ink_06_p_boundary_device_stress.py`
|
| 463 |
- `scripts/export_math_ink_06_litert.py`
|
| 464 |
- `scripts/build_math_ink_06_litert_colab_bundle.py`
|
| 465 |
+
- `scripts/benchmark_math_ink_06_composite.py`
|
| 466 |
- `scripts/preflight_math_ink_06_p_formula.py`
|
| 467 |
- `scripts/analyze_crohme_lattice_failures.py`
|
| 468 |
- `tests/test_behavior_context06.py`
|
|
|
|
| 471 |
- `research/contracts/aiflow_p_formula_v1.schema.json`
|
| 472 |
- `research/colab/AIFlow_Math_Ink_06_LiteRT.ipynb`
|
| 473 |
- `research/runs/math_ink_06_litert_colab_20260724/aiflow_math_ink_06_litert_bundle.zip`
|
| 474 |
+
- `research/runs/math_ink_06_composite_cpu_benchmark_20260724/report.json`
|
| 475 |
- `research/AIFlow-MATH-INK-0.6-BEHAVIOR-CONTEXT-REPORT-20260724.md`
|
| 476 |
- `research/runs/math_ink_06_behavior_role_3seed_20260724/run_summary.json`
|
| 477 |
- `research/runs/math_ink_06_behavior_grouping_audit_20260724/report.json`
|
reports/composite_cpu_benchmark.json
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"schema": "aiflow-math-ink-06-composite-cpu-benchmark-v1",
|
| 3 |
+
"generated_at": "2026-07-23T20:24:41.860993+00:00",
|
| 4 |
+
"torch_version": "2.5.1+cpu",
|
| 5 |
+
"platform": "win32",
|
| 6 |
+
"model_version": "aiflow-math-ink-0.6-federated-online1+aiflow-math-ink-0.6-skeleton-adapter1",
|
| 7 |
+
"model_state_bytes": 8947108,
|
| 8 |
+
"baseline_process_rss_bytes": 214175744,
|
| 9 |
+
"maximum_observed_process_rss_bytes": 231530496,
|
| 10 |
+
"inference_rss_growth_bytes": 17354752,
|
| 11 |
+
"model_state_plus_inference_growth_bytes": 26301860,
|
| 12 |
+
"memory_proxy_gate_le_100mib": true,
|
| 13 |
+
"rows": [
|
| 14 |
+
{
|
| 15 |
+
"threads": 1,
|
| 16 |
+
"online": {
|
| 17 |
+
"samples": 76,
|
| 18 |
+
"mean_ms": 6.648381579189414,
|
| 19 |
+
"p50_ms": 6.038199993781745,
|
| 20 |
+
"p95_ms": 10.202200006460771,
|
| 21 |
+
"maximum_ms": 10.744899991550483,
|
| 22 |
+
"output_checksum": 352945188,
|
| 23 |
+
"observed_process_rss_bytes": 219648000
|
| 24 |
+
},
|
| 25 |
+
"raster": {
|
| 26 |
+
"samples": 76,
|
| 27 |
+
"mean_ms": 19.39044473711922,
|
| 28 |
+
"p50_ms": 19.06099999905564,
|
| 29 |
+
"p95_ms": 20.654299994930625,
|
| 30 |
+
"maximum_ms": 28.47059999476187,
|
| 31 |
+
"output_checksum": 1642561801,
|
| 32 |
+
"observed_process_rss_bytes": 226373632
|
| 33 |
+
},
|
| 34 |
+
"proxy_gates": {
|
| 35 |
+
"online_p95_le_50ms": true,
|
| 36 |
+
"raster_p95_le_200ms": true
|
| 37 |
+
}
|
| 38 |
+
},
|
| 39 |
+
{
|
| 40 |
+
"threads": 2,
|
| 41 |
+
"online": {
|
| 42 |
+
"samples": 76,
|
| 43 |
+
"mean_ms": 6.068777630724454,
|
| 44 |
+
"p50_ms": 5.20410000171978,
|
| 45 |
+
"p95_ms": 8.508300001267344,
|
| 46 |
+
"maximum_ms": 8.986200002254918,
|
| 47 |
+
"output_checksum": 352945188,
|
| 48 |
+
"observed_process_rss_bytes": 226631680
|
| 49 |
+
},
|
| 50 |
+
"raster": {
|
| 51 |
+
"samples": 76,
|
| 52 |
+
"mean_ms": 16.508685525255522,
|
| 53 |
+
"p50_ms": 14.63520000106655,
|
| 54 |
+
"p95_ms": 23.574999999254942,
|
| 55 |
+
"maximum_ms": 24.597199997515418,
|
| 56 |
+
"output_checksum": 1642561801,
|
| 57 |
+
"observed_process_rss_bytes": 230453248
|
| 58 |
+
},
|
| 59 |
+
"proxy_gates": {
|
| 60 |
+
"online_p95_le_50ms": true,
|
| 61 |
+
"raster_p95_le_200ms": true
|
| 62 |
+
}
|
| 63 |
+
},
|
| 64 |
+
{
|
| 65 |
+
"threads": 4,
|
| 66 |
+
"online": {
|
| 67 |
+
"samples": 76,
|
| 68 |
+
"mean_ms": 6.707876315646756,
|
| 69 |
+
"p50_ms": 7.187800001702271,
|
| 70 |
+
"p95_ms": 7.77469998865854,
|
| 71 |
+
"maximum_ms": 8.105300003080629,
|
| 72 |
+
"output_checksum": 352945188,
|
| 73 |
+
"observed_process_rss_bytes": 230817792
|
| 74 |
+
},
|
| 75 |
+
"raster": {
|
| 76 |
+
"samples": 76,
|
| 77 |
+
"mean_ms": 17.300657895822567,
|
| 78 |
+
"p50_ms": 17.91409999714233,
|
| 79 |
+
"p95_ms": 19.176700006937608,
|
| 80 |
+
"maximum_ms": 20.39449999574572,
|
| 81 |
+
"output_checksum": 1642561801,
|
| 82 |
+
"observed_process_rss_bytes": 231530496
|
| 83 |
+
},
|
| 84 |
+
"proxy_gates": {
|
| 85 |
+
"online_p95_le_50ms": true,
|
| 86 |
+
"raster_p95_le_200ms": true
|
| 87 |
+
}
|
| 88 |
+
}
|
| 89 |
+
],
|
| 90 |
+
"interpretation_limit": "Windows PyTorch CPU proxy이며 Android LiteRT·배터리·delegate 성능 판정이 아니다.",
|
| 91 |
+
"product_validation": false
|
| 92 |
+
}
|
scripts/benchmark_math_ink_06_composite.py
ADDED
|
@@ -0,0 +1,183 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""정정된 Math Ink 0.6 composite online/raster CPU 지연을 대표 입력으로 측정한다."""
|
| 2 |
+
|
| 3 |
+
from __future__ import annotations
|
| 4 |
+
|
| 5 |
+
import argparse
|
| 6 |
+
from datetime import datetime, timezone
|
| 7 |
+
import json
|
| 8 |
+
import math
|
| 9 |
+
from pathlib import Path
|
| 10 |
+
import statistics
|
| 11 |
+
import sys
|
| 12 |
+
import time
|
| 13 |
+
|
| 14 |
+
import torch
|
| 15 |
+
|
| 16 |
+
PROJECT_ROOT = Path(__file__).parents[1]
|
| 17 |
+
SOURCE_ROOT = PROJECT_ROOT / "src"
|
| 18 |
+
for path in (PROJECT_ROOT, SOURCE_ROOT):
|
| 19 |
+
if str(path) not in sys.path:
|
| 20 |
+
sys.path.insert(0, str(path))
|
| 21 |
+
|
| 22 |
+
from math_grid_drawer.research.math_ink_06 import MathInk06Engine
|
| 23 |
+
from scripts.export_math_ink_06_litert import _load_representative_inputs06
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
def percentile_nearest_rank06(values: list[float], percentile: float) -> float:
|
| 27 |
+
"""필요 변수: 측정값·0~1 percentile. 작동 원리: 모바일 p95와 동일한 nearest-rank 값을 반환한다."""
|
| 28 |
+
|
| 29 |
+
if not values or not 0.0 <= percentile <= 1.0:
|
| 30 |
+
raise ValueError("percentile 입력이 유효하지 않습니다.")
|
| 31 |
+
ordered = sorted(values)
|
| 32 |
+
index = max(0, min(len(ordered) - 1, math.ceil(percentile * len(ordered)) - 1))
|
| 33 |
+
return float(ordered[index])
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
def module_state_bytes06(*modules: torch.nn.Module) -> int:
|
| 37 |
+
"""필요 변수: model·adapter module. 작동 원리: 중복 storage를 한 번만 세어 실제 tensor state bytes를 계산한다."""
|
| 38 |
+
|
| 39 |
+
seen: set[tuple[int, int]] = set()
|
| 40 |
+
total = 0
|
| 41 |
+
for module in modules:
|
| 42 |
+
for tensor in [*module.parameters(), *module.buffers()]:
|
| 43 |
+
storage = tensor.untyped_storage()
|
| 44 |
+
key = (storage.data_ptr(), storage.nbytes())
|
| 45 |
+
if key in seen:
|
| 46 |
+
continue
|
| 47 |
+
seen.add(key)
|
| 48 |
+
total += storage.nbytes()
|
| 49 |
+
return total
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
def _rss_bytes06() -> int | None:
|
| 53 |
+
"""필요 변수: 없음. 작동 원리: psutil이 있으면 현재 process RSS를 반환하고 없으면 명시적으로 결측 처리한다."""
|
| 54 |
+
|
| 55 |
+
try:
|
| 56 |
+
import psutil
|
| 57 |
+
except ImportError:
|
| 58 |
+
return None
|
| 59 |
+
return int(psutil.Process().memory_info().rss)
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
def _measure06(callable_, inputs: list[tuple[torch.Tensor, ...]], warmup: int) -> dict:
|
| 63 |
+
"""필요 변수: 고정 inference callable·대표 입력·warmup. 작동 원리: 표본별 wall latency와 output checksum을 측정한다."""
|
| 64 |
+
|
| 65 |
+
with torch.inference_mode():
|
| 66 |
+
for arguments in inputs[:max(1, min(warmup, len(inputs)))]:
|
| 67 |
+
callable_(*arguments)
|
| 68 |
+
latencies, checksum = [], 0
|
| 69 |
+
observed_rss = _rss_bytes06()
|
| 70 |
+
for arguments in inputs:
|
| 71 |
+
started = time.perf_counter()
|
| 72 |
+
output = callable_(*arguments)
|
| 73 |
+
latencies.append((time.perf_counter() - started) * 1000.0)
|
| 74 |
+
primary = output[0] if isinstance(output, tuple) else output
|
| 75 |
+
checksum = (checksum * 131 + int(primary.argmax(dim=-1)[0])) % 2_147_483_647
|
| 76 |
+
current_rss = _rss_bytes06()
|
| 77 |
+
if current_rss is not None:
|
| 78 |
+
observed_rss = max(observed_rss or 0, current_rss)
|
| 79 |
+
return {
|
| 80 |
+
"samples": len(latencies),
|
| 81 |
+
"mean_ms": statistics.fmean(latencies),
|
| 82 |
+
"p50_ms": percentile_nearest_rank06(latencies, 0.50),
|
| 83 |
+
"p95_ms": percentile_nearest_rank06(latencies, 0.95),
|
| 84 |
+
"maximum_ms": max(latencies),
|
| 85 |
+
"output_checksum": checksum,
|
| 86 |
+
"observed_process_rss_bytes": observed_rss,
|
| 87 |
+
}
|
| 88 |
+
|
| 89 |
+
|
| 90 |
+
def main() -> None:
|
| 91 |
+
"""필요 변수: composite artifact·대표 cache. 작동 원리: thread별 두 inference 경로를 독립 측정해 JSON으로 남긴다."""
|
| 92 |
+
|
| 93 |
+
parser = argparse.ArgumentParser(description="Benchmark Math Ink 0.6 composite CPU")
|
| 94 |
+
parser.add_argument("--checkpoint", type=Path, required=True)
|
| 95 |
+
parser.add_argument("--adapter-checkpoint", type=Path, required=True)
|
| 96 |
+
parser.add_argument("--representative-inputs", type=Path, required=True)
|
| 97 |
+
parser.add_argument("--threads", type=int, action="append", default=None)
|
| 98 |
+
parser.add_argument("--samples", type=int, default=76)
|
| 99 |
+
parser.add_argument("--warmup", type=int, default=5)
|
| 100 |
+
parser.add_argument("--output", type=Path, required=True)
|
| 101 |
+
args = parser.parse_args()
|
| 102 |
+
requested_threads = args.threads or [1, 2, 4]
|
| 103 |
+
if any(value <= 0 for value in requested_threads):
|
| 104 |
+
raise ValueError("CPU thread는 양수여야 합니다.")
|
| 105 |
+
torch.set_num_interop_threads(1)
|
| 106 |
+
engine = MathInk06Engine(
|
| 107 |
+
args.checkpoint, adapter_checkpoint=args.adapter_checkpoint, device="cpu",
|
| 108 |
+
)
|
| 109 |
+
online_inputs, raster_inputs = _load_representative_inputs06(args.representative_inputs)
|
| 110 |
+
online_inputs = online_inputs[:args.samples]
|
| 111 |
+
raster_inputs = raster_inputs[:args.samples]
|
| 112 |
+
|
| 113 |
+
def online_forward(sequence: torch.Tensor):
|
| 114 |
+
"""필요 변수: canonical sequence. 작동 원리: 실제 runtime online composite branch를 호출한다."""
|
| 115 |
+
|
| 116 |
+
return engine.model.forward_online(engine.online_adapter(sequence))
|
| 117 |
+
|
| 118 |
+
def raster_forward(raster: torch.Tensor):
|
| 119 |
+
"""필요 변수: raster. 작동 원리: 실제 runtime virtual stroke·raster adapter·fusion을 호출한다."""
|
| 120 |
+
|
| 121 |
+
output = engine._forward_raster_composite06(raster)
|
| 122 |
+
return engine.fuse_raster_output(output)[0]
|
| 123 |
+
|
| 124 |
+
baseline_rss = _rss_bytes06()
|
| 125 |
+
rows = []
|
| 126 |
+
for thread_count in requested_threads:
|
| 127 |
+
torch.set_num_threads(thread_count)
|
| 128 |
+
online = _measure06(online_forward, online_inputs, args.warmup)
|
| 129 |
+
raster = _measure06(raster_forward, raster_inputs, args.warmup)
|
| 130 |
+
rows.append({
|
| 131 |
+
"threads": thread_count,
|
| 132 |
+
"online": online,
|
| 133 |
+
"raster": raster,
|
| 134 |
+
"proxy_gates": {
|
| 135 |
+
"online_p95_le_50ms": online["p95_ms"] <= 50.0,
|
| 136 |
+
"raster_p95_le_200ms": raster["p95_ms"] <= 200.0,
|
| 137 |
+
},
|
| 138 |
+
})
|
| 139 |
+
observed_rss_values = [
|
| 140 |
+
int(metrics["observed_process_rss_bytes"])
|
| 141 |
+
for row in rows for metrics in (row["online"], row["raster"])
|
| 142 |
+
if metrics["observed_process_rss_bytes"] is not None
|
| 143 |
+
]
|
| 144 |
+
maximum_observed_rss = max(observed_rss_values) if observed_rss_values else None
|
| 145 |
+
inference_rss_growth = (
|
| 146 |
+
max(0, maximum_observed_rss - baseline_rss)
|
| 147 |
+
if maximum_observed_rss is not None and baseline_rss is not None else None
|
| 148 |
+
)
|
| 149 |
+
state_bytes = module_state_bytes06(engine.model, engine.composite_adapter)
|
| 150 |
+
model_plus_inference = (
|
| 151 |
+
state_bytes + inference_rss_growth
|
| 152 |
+
if inference_rss_growth is not None else None
|
| 153 |
+
)
|
| 154 |
+
report = {
|
| 155 |
+
"schema": "aiflow-math-ink-06-composite-cpu-benchmark-v1",
|
| 156 |
+
"generated_at": datetime.now(timezone.utc).isoformat(),
|
| 157 |
+
"torch_version": torch.__version__,
|
| 158 |
+
"platform": sys.platform,
|
| 159 |
+
"model_version": engine.model_version,
|
| 160 |
+
"model_state_bytes": state_bytes,
|
| 161 |
+
"baseline_process_rss_bytes": baseline_rss,
|
| 162 |
+
"maximum_observed_process_rss_bytes": maximum_observed_rss,
|
| 163 |
+
"inference_rss_growth_bytes": inference_rss_growth,
|
| 164 |
+
"model_state_plus_inference_growth_bytes": model_plus_inference,
|
| 165 |
+
"memory_proxy_gate_le_100mib": (
|
| 166 |
+
model_plus_inference <= 100 * 1024 * 1024
|
| 167 |
+
if model_plus_inference is not None else None
|
| 168 |
+
),
|
| 169 |
+
"rows": rows,
|
| 170 |
+
"interpretation_limit": (
|
| 171 |
+
"Windows PyTorch CPU proxy이며 Android LiteRT·배터리·delegate 성능 판정이 아니다."
|
| 172 |
+
),
|
| 173 |
+
"product_validation": False,
|
| 174 |
+
}
|
| 175 |
+
args.output.parent.mkdir(parents=True, exist_ok=True)
|
| 176 |
+
args.output.write_text(
|
| 177 |
+
json.dumps(report, ensure_ascii=False, indent=2) + "\n", encoding="utf-8",
|
| 178 |
+
)
|
| 179 |
+
print(json.dumps(report, ensure_ascii=False, indent=2))
|
| 180 |
+
|
| 181 |
+
|
| 182 |
+
if __name__ == "__main__":
|
| 183 |
+
main()
|