{ "experiment": "R-MATH-INK-06-BEHAVIOR-ROLE-001", "generated_at": "2026-07-23T16:42:37.095745+00:00", "seed": 47, "device": "cuda", "cuda_device": "NVIDIA GeForce GTX 1650", "teacher_adapter": "research\\runs\\math_ink_06_online_casecontext_refined_seed47_20260723\\skeleton_adapter.pt", "base_checkpoint": "C:\\Users\\user\\Desktop\\Aiflow\\math-grid-drawer-simulation\\research\\runs\\math_ink_06_federated_virtual_ce025_family010_seed47_20260723\\math_ink_06_candidate.pt", "split_contract": "CROHME2012 trainData writer fit/validation; testDataGT official held-out", "formulas": { "fit": 1063, "validation": 275, "official_test": 488 }, "role_samples": { "fit": { "identifier_lower": 1573, "multiply_operator": 177, "identifier_upper": 99 }, "validation": { "identifier_lower": 401, "multiply_operator": 44, "identifier_upper": 22 }, "official_test": { "identifier_lower": 568, "identifier_upper": 31, "multiply_operator": 36 } }, "teacher_baseline": { "validation": { "samples": 467, "accuracy": 0.4368308484554291, "macro_f1": 0.24164757569154482, "recall": { "identifier_lower": 0.4713216957605985, "identifier_upper": 0.6818181818181818, "multiply_operator": 0.0 }, "confusion": [ [ 189, 212, 0 ], [ 7, 15, 0 ], [ 16, 28, 0 ] ], "ece": 0.3735619432834072 }, "official_test": { "samples": 635, "accuracy": 0.6094487905502319, "macro_f1": 0.31121831754743146, "recall": { "identifier_lower": 0.6390845070422535, "identifier_upper": 0.7741935483870968, "multiply_operator": 0.0 }, "confusion": [ [ 363, 205, 0 ], [ 7, 24, 0 ], [ 10, 26, 0 ] ], "ece": 0.1850307746969806 } }, "selected_epoch": 22, "validation": { "samples": 467, "accuracy": 0.9892933368682861, "macro_f1": 0.9606228956228956, "recall": { "identifier_lower": 0.9925187032418953, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 398, 2, 1 ], [ 1, 20, 1 ], [ 0, 0, 44 ] ], "ece": 0.012289993044929143 }, "official_test": { "samples": 635, "accuracy": 0.930708646774292, "macro_f1": 0.7541782507773837, "recall": { "identifier_lower": 0.9559859154929577, "identifier_upper": 0.5483870967741935, "multiply_operator": 0.8611111111111112 }, "confusion": [ [ 543, 11, 14 ], [ 9, 17, 5 ], [ 5, 0, 31 ] ], "ece": 0.048238128075605236 }, "history": [ { "epoch": 1, "training_loss": 0.9046294994777186, "learning_rate": 0.0009993147673772868, "validation": { "samples": 467, "accuracy": 0.9229121804237366, "macro_f1": 0.6002162746688294, "recall": { "identifier_lower": 0.9800498753117207, "identifier_upper": 0.0, "multiply_operator": 0.8636363636363636 }, "confusion": [ [ 393, 0, 8 ], [ 22, 0, 0 ], [ 6, 0, 38 ] ], "ece": 0.2843504021118433 } }, { "epoch": 2, "training_loss": 0.5676413621690997, "learning_rate": 0.0009972609476841365, "validation": { "samples": 467, "accuracy": 0.9443255066871643, "macro_f1": 0.8217141861204237, "recall": { "identifier_lower": 0.9600997506234414, "identifier_upper": 0.5909090909090909, "multiply_operator": 0.9772727272727273 }, "confusion": [ [ 385, 6, 10 ], [ 6, 13, 3 ], [ 1, 0, 43 ] ], "ece": 0.21609727580390234 } }, { "epoch": 3, "training_loss": 0.42114428641539514, "learning_rate": 0.0009938441702975688, "validation": { "samples": 467, "accuracy": 0.9293361902236938, "macro_f1": 0.8249043707409466, "recall": { "identifier_lower": 0.9226932668329177, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 370, 21, 10 ], [ 0, 20, 2 ], [ 0, 0, 44 ] ], "ece": 0.14836443936291555 } }, { "epoch": 4, "training_loss": 0.332364505887225, "learning_rate": 0.0009890738003669028, "validation": { "samples": 467, "accuracy": 0.9421841502189636, "macro_f1": 0.8434321608170148, "recall": { "identifier_lower": 0.940149625935162, "identifier_upper": 0.8636363636363636, "multiply_operator": 1.0 }, "confusion": [ [ 377, 17, 7 ], [ 1, 19, 2 ], [ 0, 0, 44 ] ], "ece": 0.10876856215048941 } }, { "epoch": 5, "training_loss": 0.2625855236939316, "learning_rate": 0.0009829629131445341, "validation": { "samples": 467, "accuracy": 0.9635974168777466, "macro_f1": 0.8965055628931521, "recall": { "identifier_lower": 0.9600997506234414, "identifier_upper": 0.9545454545454546, "multiply_operator": 1.0 }, "confusion": [ [ 385, 12, 4 ], [ 0, 21, 1 ], [ 0, 0, 44 ] ], "ece": 0.0713321801207063 } }, { "epoch": 6, "training_loss": 0.21212876709394676, "learning_rate": 0.0009755282581475769, "validation": { "samples": 467, "accuracy": 0.9657387733459473, "macro_f1": 0.9016515388200431, "recall": { "identifier_lower": 0.9625935162094763, "identifier_upper": 0.9545454545454546, "multiply_operator": 1.0 }, "confusion": [ [ 386, 11, 4 ], [ 0, 21, 1 ], [ 0, 0, 44 ] ], "ece": 0.05290893672154348 } }, { "epoch": 7, "training_loss": 0.17808458503595617, "learning_rate": 0.0009667902132486009, "validation": { "samples": 467, "accuracy": 0.9764453768730164, "macro_f1": 0.9259787053904701, "recall": { "identifier_lower": 0.9750623441396509, "identifier_upper": 0.9545454545454546, "multiply_operator": 1.0 }, "confusion": [ [ 391, 8, 2 ], [ 0, 21, 1 ], [ 0, 0, 44 ] ], "ece": 0.046399454602882084 } }, { "epoch": 8, "training_loss": 0.1623947636832283, "learning_rate": 0.0009567727288213005, "validation": { "samples": 467, "accuracy": 0.9721627235412598, "macro_f1": 0.9147653079346415, "recall": { "identifier_lower": 0.970074812967581, "identifier_upper": 0.9545454545454546, "multiply_operator": 1.0 }, "confusion": [ [ 389, 10, 2 ], [ 0, 21, 1 ], [ 0, 0, 44 ] ], "ece": 0.04120317458101366 } }, { "epoch": 9, "training_loss": 0.13445646132268474, "learning_rate": 0.0009455032620941839, "validation": { "samples": 467, "accuracy": 0.9700214266777039, "macro_f1": 0.9105244977343826, "recall": { "identifier_lower": 0.970074812967581, "identifier_upper": 0.9545454545454546, "multiply_operator": 0.9772727272727273 }, "confusion": [ [ 389, 10, 2 ], [ 0, 21, 1 ], [ 1, 0, 43 ] ], "ece": 0.024122650888779656 } }, { "epoch": 10, "training_loss": 0.12278934674948985, "learning_rate": 0.0009330127018922195, "validation": { "samples": 467, "accuracy": 0.978586733341217, "macro_f1": 0.934105096614006, "recall": { "identifier_lower": 0.9775561097256857, "identifier_upper": 0.9545454545454546, "multiply_operator": 1.0 }, "confusion": [ [ 392, 6, 3 ], [ 0, 21, 1 ], [ 0, 0, 44 ] ], "ece": 0.03280332642193773 } }, { "epoch": 11, "training_loss": 0.11912762667034818, "learning_rate": 0.0009193352839727121, "validation": { "samples": 467, "accuracy": 0.9743040800094604, "macro_f1": 0.9202729423968362, "recall": { "identifier_lower": 0.972568578553616, "identifier_upper": 0.9545454545454546, "multiply_operator": 1.0 }, "confusion": [ [ 390, 9, 2 ], [ 0, 21, 1 ], [ 0, 0, 44 ] ], "ece": 0.02059314902723597 } }, { "epoch": 12, "training_loss": 0.1007714251961948, "learning_rate": 0.0009045084971874737, "validation": { "samples": 467, "accuracy": 0.9764453768730164, "macro_f1": 0.9278519869239424, "recall": { "identifier_lower": 0.9800498753117207, "identifier_upper": 0.9090909090909091, "multiply_operator": 0.9772727272727273 }, "confusion": [ [ 393, 5, 3 ], [ 1, 20, 1 ], [ 1, 0, 43 ] ], "ece": 0.028532721249018206 } }, { "epoch": 13, "training_loss": 0.09389568820974516, "learning_rate": 0.0008885729807284855, "validation": { "samples": 467, "accuracy": 0.9764453768730164, "macro_f1": 0.9259787053904701, "recall": { "identifier_lower": 0.9750623441396509, "identifier_upper": 0.9545454545454546, "multiply_operator": 1.0 }, "confusion": [ [ 391, 8, 2 ], [ 0, 21, 1 ], [ 0, 0, 44 ] ], "ece": 0.015950765151738794 } }, { "epoch": 14, "training_loss": 0.09023109214202077, "learning_rate": 0.0008715724127386972, "validation": { "samples": 467, "accuracy": 0.978586733341217, "macro_f1": 0.9296818485497731, "recall": { "identifier_lower": 0.9800498753117207, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 393, 6, 2 ], [ 1, 20, 1 ], [ 0, 0, 44 ] ], "ece": 0.017414743734838506 } }, { "epoch": 15, "training_loss": 0.08063163422954733, "learning_rate": 0.0008535533905932739, "validation": { "samples": 467, "accuracy": 0.978586733341217, "macro_f1": 0.9318945535338977, "recall": { "identifier_lower": 0.9775561097256857, "identifier_upper": 0.9545454545454546, "multiply_operator": 1.0 }, "confusion": [ [ 392, 7, 2 ], [ 0, 21, 1 ], [ 0, 0, 44 ] ], "ece": 0.01352588653162315 } }, { "epoch": 16, "training_loss": 0.08132128170482399, "learning_rate": 0.0008345653031794292, "validation": { "samples": 467, "accuracy": 0.978586733341217, "macro_f1": 0.9318945535338977, "recall": { "identifier_lower": 0.9775561097256857, "identifier_upper": 0.9545454545454546, "multiply_operator": 1.0 }, "confusion": [ [ 392, 7, 2 ], [ 0, 21, 1 ], [ 0, 0, 44 ] ], "ece": 0.009898379995308015 } }, { "epoch": 17, "training_loss": 0.06615571991584313, "learning_rate": 0.0008146601955249188, "validation": { "samples": 467, "accuracy": 0.9828693866729736, "macro_f1": 0.9400195571849914, "recall": { "identifier_lower": 0.9850374064837906, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 395, 5, 1 ], [ 1, 20, 1 ], [ 0, 0, 44 ] ], "ece": 0.013311798741971193 } }, { "epoch": 18, "training_loss": 0.06652265662209546, "learning_rate": 0.0007938926261462366, "validation": { "samples": 467, "accuracy": 0.980728030204773, "macro_f1": 0.9329268458237093, "recall": { "identifier_lower": 0.9825436408977556, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 394, 5, 2 ], [ 0, 20, 2 ], [ 0, 0, 44 ] ], "ece": 0.012934121393568976 } }, { "epoch": 19, "training_loss": 0.06063114960913145, "learning_rate": 0.0007723195175075136, "validation": { "samples": 467, "accuracy": 0.978586733341217, "macro_f1": 0.9290749407254631, "recall": { "identifier_lower": 0.9800498753117207, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 393, 5, 3 ], [ 0, 20, 2 ], [ 0, 0, 44 ] ], "ece": 0.012781770980824225 } }, { "epoch": 20, "training_loss": 0.05799441609537685, "learning_rate": 0.00075, "validation": { "samples": 467, "accuracy": 0.9871520400047302, "macro_f1": 0.9550404244443061, "recall": { "identifier_lower": 0.9875311720698254, "identifier_upper": 0.9545454545454546, "multiply_operator": 1.0 }, "confusion": [ [ 396, 4, 1 ], [ 0, 21, 1 ], [ 0, 0, 44 ] ], "ece": 0.01620049691104486 } }, { "epoch": 21, "training_loss": 0.05703890749962024, "learning_rate": 0.0007269952498697733, "validation": { "samples": 467, "accuracy": 0.9764453768730164, "macro_f1": 0.9233519504577847, "recall": { "identifier_lower": 0.9800498753117207, "identifier_upper": 0.9090909090909091, "multiply_operator": 0.9772727272727273 }, "confusion": [ [ 393, 7, 1 ], [ 1, 20, 1 ], [ 1, 0, 43 ] ], "ece": 0.009044911882096912 } }, { "epoch": 22, "training_loss": 0.05246292020502994, "learning_rate": 0.0007033683215379002, "validation": { "samples": 467, "accuracy": 0.9892933368682861, "macro_f1": 0.9606228956228956, "recall": { "identifier_lower": 0.9925187032418953, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 398, 2, 1 ], [ 1, 20, 1 ], [ 0, 0, 44 ] ], "ece": 0.012289993044929143 } }, { "epoch": 23, "training_loss": 0.05485564452561254, "learning_rate": 0.0006791839747726503, "validation": { "samples": 467, "accuracy": 0.9828693866729736, "macro_f1": 0.9368530361259967, "recall": { "identifier_lower": 0.9850374064837906, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 395, 5, 1 ], [ 0, 20, 2 ], [ 0, 0, 44 ] ], "ece": 0.008606824696161752 } }, { "epoch": 24, "training_loss": 0.05374844572338109, "learning_rate": 0.0006545084971874737, "validation": { "samples": 467, "accuracy": 0.980728030204773, "macro_f1": 0.9359007370090494, "recall": { "identifier_lower": 0.9800498753117207, "identifier_upper": 0.9545454545454546, "multiply_operator": 1.0 }, "confusion": [ [ 393, 7, 1 ], [ 0, 21, 1 ], [ 0, 0, 44 ] ], "ece": 0.00825181347225945 } }, { "epoch": 25, "training_loss": 0.041329645392071436, "learning_rate": 0.0006294095225512603, "validation": { "samples": 467, "accuracy": 0.9850106835365295, "macro_f1": 0.946608066058867, "recall": { "identifier_lower": 0.9875311720698254, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 396, 4, 1 ], [ 1, 20, 1 ], [ 0, 0, 44 ] ], "ece": 0.01209575774203274 } }, { "epoch": 26, "training_loss": 0.04697309928835889, "learning_rate": 0.0006039558454088795, "validation": { "samples": 467, "accuracy": 0.9828693866729736, "macro_f1": 0.9400195571849914, "recall": { "identifier_lower": 0.9850374064837906, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 395, 5, 1 ], [ 1, 20, 1 ], [ 0, 0, 44 ] ], "ece": 0.010180474722224608 } }, { "epoch": 27, "training_loss": 0.03527613957611499, "learning_rate": 0.0005782172325201154, "validation": { "samples": 467, "accuracy": 0.9828693866729736, "macro_f1": 0.9400195571849914, "recall": { "identifier_lower": 0.9850374064837906, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 395, 5, 1 ], [ 1, 20, 1 ], [ 0, 0, 44 ] ], "ece": 0.007127549435960195 } }, { "epoch": 28, "training_loss": 0.038951482444501945, "learning_rate": 0.0005522642316338267, "validation": { "samples": 467, "accuracy": 0.9828693866729736, "macro_f1": 0.9400195571849914, "recall": { "identifier_lower": 0.9850374064837906, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 395, 5, 1 ], [ 1, 20, 1 ], [ 0, 0, 44 ] ], "ece": 0.004555291684447127 } }, { "epoch": 29, "training_loss": 0.0438767217397851, "learning_rate": 0.0005261679781214719, "validation": { "samples": 467, "accuracy": 0.978586733341217, "macro_f1": 0.9275945178910138, "recall": { "identifier_lower": 0.9800498753117207, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 393, 7, 1 ], [ 1, 20, 1 ], [ 0, 0, 44 ] ], "ece": 0.012363891069576471 } }, { "epoch": 30, "training_loss": 0.03519111297512906, "learning_rate": 0.0005000000000000001, "validation": { "samples": 467, "accuracy": 0.980728030204773, "macro_f1": 0.9357769673206845, "recall": { "identifier_lower": 0.9850374064837906, "identifier_upper": 0.9090909090909091, "multiply_operator": 0.9772727272727273 }, "confusion": [ [ 395, 5, 1 ], [ 1, 20, 1 ], [ 1, 0, 43 ] ], "ece": 0.013340822110229938 } }, { "epoch": 31, "training_loss": 0.0424999847520584, "learning_rate": 0.00047383202187852816, "validation": { "samples": 467, "accuracy": 0.980728030204773, "macro_f1": 0.9336869532849432, "recall": { "identifier_lower": 0.9825436408977556, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 394, 6, 1 ], [ 1, 20, 1 ], [ 0, 0, 44 ] ], "ece": 0.011168826406788984 } }, { "epoch": 32, "training_loss": 0.03117308776328021, "learning_rate": 0.00044773576836617336, "validation": { "samples": 467, "accuracy": 0.9871520400047302, "macro_f1": 0.9534696147962731, "recall": { "identifier_lower": 0.9900249376558603, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 397, 3, 1 ], [ 1, 20, 1 ], [ 0, 0, 44 ] ], "ece": 0.005621013650850271 } }, { "epoch": 33, "training_loss": 0.03737180095756292, "learning_rate": 0.0004217827674798846, "validation": { "samples": 467, "accuracy": 0.9871520400047302, "macro_f1": 0.9534696147962731, "recall": { "identifier_lower": 0.9900249376558603, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 397, 3, 1 ], [ 1, 20, 1 ], [ 0, 0, 44 ] ], "ece": 0.011075654017887793 } }, { "epoch": 34, "training_loss": 0.03798794678300055, "learning_rate": 0.00039604415459112036, "validation": { "samples": 467, "accuracy": 0.980728030204773, "macro_f1": 0.9336869532849432, "recall": { "identifier_lower": 0.9825436408977556, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 394, 6, 1 ], [ 1, 20, 1 ], [ 0, 0, 44 ] ], "ece": 0.010551084664118249 } }, { "epoch": 35, "training_loss": 0.03510260305746689, "learning_rate": 0.00037059047744873974, "validation": { "samples": 467, "accuracy": 0.9828693866729736, "macro_f1": 0.9400195571849914, "recall": { "identifier_lower": 0.9850374064837906, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 395, 5, 1 ], [ 1, 20, 1 ], [ 0, 0, 44 ] ], "ece": 0.008203305500332658 } }, { "epoch": 36, "training_loss": 0.03304981503569158, "learning_rate": 0.0003454915028125264, "validation": { "samples": 467, "accuracy": 0.9828693866729736, "macro_f1": 0.9400195571849914, "recall": { "identifier_lower": 0.9850374064837906, "identifier_upper": 0.9090909090909091, "multiply_operator": 1.0 }, "confusion": [ [ 395, 5, 1 ], [ 1, 20, 1 ], [ 0, 0, 44 ] ], "ece": 0.0067995716398946415 } } ], "checkpoint": "behavior_role_head.pt", "checkpoint_bytes": 75794, "track": "R_noncommercial_only", "product_validation": false, "interpretation_limit": "CROHME 연구용 연속 수식 행동 head이며 상용 checkpoint에 병합할 수 없다. P-track writer/device-disjoint 재학습이 필요하다." }