{ "results": [ { "theta": 160000.0, "short": 0.9160820834562606, "mid": 1.0387379115148965, "long": 1.1367697905727685, "seconds": 26.022907972335815 }, { "theta": 640000.0, "short": 0.9236352910098977, "mid": 1.0243582937320355, "long": 0.9822827437326973, "seconds": 11.011847972869873 }, { "theta": 1280000.0, "short": 0.9304989230334977, "mid": 1.031460880067349, "long": 0.9601743056856983, "seconds": 11.087600231170654 }, { "theta": 2560000.0, "short": 0.939639166783192, "mid": 1.044918367936274, "long": 0.9650228895573634, "seconds": 11.14451551437378 }, { "theta": 5120000.0, "short": 0.9498286157172525, "mid": 1.0788166451562344, "long": 1.0147006089168509, "seconds": 11.193717956542969 }, { "theta": 10240000.0, "short": 0.9709671184858385, "mid": 1.0868880858439343, "long": 0.9965657126456587, "seconds": 11.231298685073853 } ], "chosen_theta": 1280000.0, "samples": { "short": 256, "mid": 96, "long": 64 }, "rule": "min(mid + long masked-LM loss) with short-context loss within 5% of the original 160k theta; 15% masking; training text only" }