{"id": "26b636bbe8044bcca5545097ec8c68d260d12a718a2afe49171bfae062457183:action", "state_id": "667f011ce3742ed32da8c60947a4f6d71836360951e78f8586899f6b71428c55", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.55, 0.45], "teacher_probs": [0.55, 0.45], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.92578125, -1.9921875], "student_probs": [0.5165954232215881, 0.4834045171737671], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.55, 0.45], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6a9486711a4e9cadb459aaa14a609c158e315beef64baf27d60b3d7fe7796403:action", "state_id": "561ef3e949655ad893cc8401d06e7b9226ac8684686b7d9aa77c67773eb0b332", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.26, 0.35, 0.39], "teacher_probs": [0.26, 0.35, 0.39], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.12890625, -2.25, -2.28515625], "student_probs": [0.3647909164428711, 0.32318684458732605, 0.31202220916748047], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.26, 0.35, 0.39], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0f5c4fb33eb2e27f9da723e967110bb512d7fcdeff53cdade491582f78602ba9:action", "state_id": "3fd63bd2d4885b520eb5302a3defb9a0d0962f68424f3f5a9e189bcfcbeec355", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.18, 0.61, 0.21], "teacher_probs": [0.18, 0.61, 0.21], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.19140625, -2.21484375, -2.7734375], "student_probs": [0.1932671219110489, 0.513184666633606, 0.29354825615882874], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.18, 0.61, 0.21], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7826348e1406a2db89b8f15f6f333c81fdbae8c2d72145652179c71b80b832c7:action", "state_id": "789e4a733ebfcdc02c5020c86dbdb62f7096d6bda5aa1cde45a3ea82f11ecff7", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.27, 0.73], "teacher_probs": [0.27, 0.73], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.52734375, -2.13671875], "student_probs": [0.4035668671131134, 0.596433162689209], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.27, 0.73], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6790f406be68b167b9e2316325cc5ccb0e390995f07d62319eda31aa61482805:action", "state_id": "0b9c2f691071bde36a41b2063a24044bebcd2aba62129cd7574aea0e94eb5acb", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.28, 0.72], "teacher_probs": [0.28, 0.72], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.07421875, -2.359375], "student_probs": [0.3285294473171234, 0.671470582485199], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.28, 0.72], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cbceb63470cb219c8e0e4c69a8da5085d9a1d37d9d769aa1cd6dbf6847fa68a6:action", "state_id": "c6adf8a6d16a011d445a29116ade618cf9bff3e1905b968b16a1b817d25fc51a", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.4, 0.26, 0.34], "teacher_probs": [0.4, 0.26, 0.34], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.48828125, -2.0859375, -1.3671875], "student_probs": [0.3732972741127014, 0.2053506076335907, 0.4213520884513855], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.4, 0.26, 0.34], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d7bb918b7e0548f8e5e1b74012c12e8a9906c747f4f141fb872f57d95429c0c9:action", "state_id": "9ce1acea8930b500911794ace273b1deff1a1d7df3356b3f88d50eb844c440b4", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.42, 0.21, 0.37], "teacher_probs": [0.42, 0.21, 0.37], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.3359375, -2.08203125, -0.33203125], "student_probs": [0.23791758716106415, 0.11282417178153992, 0.6492582559585571], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.42, 0.21, 0.37], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6d73b31b503cc576545cdd0af0bbbd238a3f4dcda10cfc40b11f3745b83ea568:action", "state_id": "28abb4f24df8b1d5a241559af132e4c5fbb1d2d2ea1a466dbb0b69aff7ac80b1", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.42, 0.23, 0.35], "teacher_probs": [0.42, 0.23, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.951171875, -2.01953125, -0.670166015625], "student_probs": [0.3748079836368561, 0.12877342104911804, 0.49641868472099304], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.42, 0.23, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "09af1295368b0e4ff34c725c330a7de09776f89b295116db11ebd4ba44945ecb:action", "state_id": "539c550f59698d16eb48e878df231d169821d6ba9d94203056aadc3850bbe9c1", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.49, 0.51], "teacher_probs": [0.49, 0.51], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.91015625, -0.70703125], "student_probs": [0.2309197634458542, 0.7690802216529846], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.51], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3ed760179cf49e61a84a1403e0f05a1e0f9659ac1652694019601eb5cf5016b3:action", "state_id": "e43cc78621511c75bd20a69a6968fb56bfec9b8f8b3f806cf173c4b60289b552", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.55, 0.45], "teacher_probs": [0.55, 0.45], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.47265625, -2.13671875], "student_probs": [0.6601723432540894, 0.33982759714126587], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.55, 0.45], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5632a7e31e17fc3635d2e2b19d8f03a80cd05c412cfa970cdd463e8a31b63a2f:action", "state_id": "4e507db3aa5b35afc8b0f45fbf59da09c17a83da91c1f7d72bfe1f84f651fd28", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "west"], "gold_index": null, "teacher_raw_probs": [0.49, 0.51], "teacher_probs": [0.49, 0.51], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.32421875, -1.7265625], "student_probs": [0.5992506742477417, 0.4007493853569031], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.51], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ae9c15a01ae1df77d4cdf3ce9621f225db7d96e74e7abb4c6620dab2e5f5a675:action", "state_id": "aee221509f1db738da34189ab83f3f091e642358e5b0d0f2456ee30e0875c1a1", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.44, 0.18, 0.38], "teacher_probs": [0.44, 0.18, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.33984375, -1.70703125, -1.30859375], "student_probs": [0.3670501708984375, 0.2542482316493988, 0.3787015974521637], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.44, 0.18, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f300da65c11d3f30631e6959c39fbe3d6b3454aff758d74bf5f7a4843baf0583:action", "state_id": "618dd217ec01f71e70b5a4c66ea926ef1916715af511e89d2c5bf99a457637a3", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.21, 0.2, 0.59], "teacher_probs": [0.21, 0.2, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.32421875, -2.5078125, -1.66796875], "student_probs": [0.2659698724746704, 0.2213597595691681, 0.5126703381538391], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.21, 0.2, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7a8d8480f165a5b47e89d17e28c661a216671fff83f3ffc7ae2fd4bf3800f3ef:action", "state_id": "c5092746493aaf17ec0a33d026b09a03d35c499eee1d0e2c096df88fd81406fb", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.5, 0.5], "teacher_probs": [0.5, 0.5], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.30078125, -1.23828125], "student_probs": [0.4843800961971283, 0.5156199336051941], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.5, 0.5], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "075c0501ea64024c88682443b2a2f0e5a5bc2794a3f2bc66f93d0fad5f417ba9:action", "state_id": "223f7ef9a11f41107e5fe726a44ee17ba06e366a962611892e9a4550427aecb3", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.41, 0.37, 0.22], "teacher_probs": [0.41, 0.37, 0.22], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.099609375, -1.546875, -2.35546875], "student_probs": [0.5196951627731323, 0.33227959275245667, 0.1480252742767334], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.37, 0.22], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "071a5949ba1cfe58ad07da5c3763c245d274620202d2b6ee5b846cbaab847878:action", "state_id": "05c76fd23cf6ff2139af18bd0fcd5dc0d8e7a73df715f4157375c85cbb6d8dfa", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.42, 0.41, 0.17], "teacher_probs": [0.42, 0.41, 0.17], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.9052734375, -1.26171875, -2.0546875], "student_probs": [0.4957900047302246, 0.34713268280029297, 0.157077357172966], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.42, 0.41, 0.17], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0d7775a511b65e7cbefe3666c7ae8a41ef2d7430d1ef609ca4a9f37a9136cfd1:action", "state_id": "6b9b24430e5ac31859e47f3ed2f88751788c9b97fb467174aa0ac36edcfdf480", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.22, 0.26, 0.52], "teacher_probs": [0.22, 0.26, 0.52], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.3125, -2.25390625, -2.25390625], "student_probs": [0.3204420506954193, 0.33977892994880676, 0.33977892994880676], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.22, 0.26, 0.52], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cc2b14a0b4a1289582628e3363854ad5b7b1b59e210e3fc8a77440de52780f9a:action", "state_id": "9ef4e39a7dac1a9d2ccb931f44e0bc403199e1363730273f7941afa2754f465a", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.19, 0.26, 0.55], "teacher_probs": [0.19, 0.26, 0.55], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.50390625, -2.23828125, -1.40234375], "student_probs": [0.18821369111537933, 0.24547693133354187, 0.5663093328475952], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.19, 0.26, 0.55], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "44436ebacc5291daa76d2a9906cee4f92705a10ff1f21dd0b767d0a8e2dc3dc3:action", "state_id": "96c5862d8d3dd7006640d0f7250555a10fc4da33a16c63c48d5f32c00ab9a1e8", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.43, 0.57], "teacher_probs": [0.43, 0.57], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.4140625, -2.1796875], "student_probs": [0.44167301058769226, 0.5583270192146301], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.43, 0.57], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7e31df98ba02a1b90ec76780b58389e90f1406e26ecbd6c2f0b0e67c33d413a4:action", "state_id": "2525d831349d98854e6686be605a264c5f7649a9bc3aff46242721b4d192336c", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.11, 0.28, 0.61], "teacher_probs": [0.11, 0.28, 0.61], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.62890625, -1.609375, -1.26953125], "student_probs": [0.1304520219564438, 0.36159929633140564, 0.5079486966133118], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.11, 0.28, 0.61], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e22c8e89cca42bd3b435bd66dfacd1e0f0c62859b0076792e4250d04d4560012:action", "state_id": "298289bbe73f5a1bffad2bc6b923a37311e58a73e336e0d278ac7c19e078fdce", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.43, 0.57], "teacher_probs": [0.43, 0.57], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.0146484375, -1.40234375], "student_probs": [0.5957277417182922, 0.4042722284793854], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.43, 0.57], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0775532cdf2f9f8e130bc14b09a883c6531783dd584ddaa3ef6c9a4af31eb8a7:action", "state_id": "b8ab4631c20de60068d0388f1fb6adcfba620fd4ef79aff74cce34b3add3c4c2", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.59, 0.41], "teacher_probs": [0.59, 0.41], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.5, -1.62109375], "student_probs": [0.5302364826202393, 0.46976348757743835], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.59, 0.41], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "48bc99e7bd7c126666bca800eb8241c732def46995193e2da3923459f38270f3:action", "state_id": "664a340cafe0f00c810d9910db99551d3954440fc2a022f332565835ad36961d", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.55, 0.45], "teacher_probs": [0.55, 0.45], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.046875, -0.9765625], "student_probs": [0.735641598701477, 0.2643583416938782], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.55, 0.45], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9c22c34e818a226f07c8e999f3d8dd49a93739f9759b26506664f7a357236c7f:action", "state_id": "07bf0e303dc44217769eb1bed6a378fd4e886adfab48d2922a2e056b87ac2aa9", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.22, 0.56, 0.22], "teacher_probs": [0.22, 0.56, 0.22], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.705810546875, 0.0078125, -0.64990234375], "student_probs": [0.24396944046020508, 0.4980328381061554, 0.2579978108406067], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.22, 0.56, 0.22], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6f03ca968630447301e3fa58e4e5850977f32ffb6917e0d3baa82a2ee08a2db1:action", "state_id": "8f59bc1fa6045154fade628d39c9d6cf1f054a2f62fe8baef99db0ca9f71c081", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.44, 0.26, 0.3], "teacher_probs": [0.44, 0.26, 0.3], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.4609375, -1.05859375, -1.74609375], "student_probs": [0.3079555630683899, 0.4604937434196472, 0.23155079782009125], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.44, 0.26, 0.3], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2cadad591859cc865ff3ed8b349ed64b0bab455c38895884a211edb5b20d78f5:action", "state_id": "fc89763a74bc84f168eadc4c35167e9be986ba388f50ef7a61b21c0bd87d6597", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.58, 0.34, 0.08], "teacher_probs": [0.58, 0.34, 0.08], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.7890625, -1.26171875, -2.1796875], "student_probs": [0.29664263129234314, 0.5026388764381409, 0.2007184624671936], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.58, 0.34, 0.08], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1cfa598831914c0975330d47f69039b92c5a94b6947f1174d54c01c52aee2766:action", "state_id": "f595444677094c747d71cadbb0807a2f7d556973e8d45f044a892dad945d5a00", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.85, 0.15], "teacher_probs": [0.85, 0.15], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.9296875, -2.0546875], "student_probs": [0.7549149990081787, 0.24508501589298248], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.85, 0.15], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "62247c02b43b4e13e5ac094f373851a1561f6b60b5ee8b74cfdffabb3576c3e1:action", "state_id": "94919a7bb01c61fe8d3a5c523a6cee20311a3eaa802329249b8569a1b046d5ba", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.67, 0.33], "teacher_probs": [0.67, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.94921875, -2.01171875], "student_probs": [0.7431679368019104, 0.25683197379112244], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.67, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d756311bf3cc2daf32ed3332c552f73ca81c5e508df0cf1677c46196193f28ab:action", "state_id": "23b579844d6e80ab61c3151d89e748f14c9d1c351c89e60133fb2e7a3d576d0e", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.39, 0.61], "teacher_probs": [0.39, 0.61], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.91796875, -1.5078125], "student_probs": [0.3988746702671051, 0.6011253595352173], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.39, 0.61], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ba8e968bf26d05e99aa17c6c15ce787003ba30ec711756ca8540a88ff6ccd5b1:action", "state_id": "e55bf55ce09d77acc288d2bec9d2659cf783fc511f1a71a0dff9f8d90f735184", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.33, 0.26, 0.41000000000000003], "teacher_probs": [0.33, 0.26, 0.41000000000000003], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.58203125, -2.0703125, -1.10546875], "student_probs": [0.3101535737514496, 0.19033512473106384, 0.4995112419128418], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.33, 0.26, 0.41000000000000003], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "00407dce8a5bed22e58bfd3c14c28031d3a05fc9237862badccd79251d4bb381:action", "state_id": "e1ea606305c61288495616a66021579b70b3e1791433f0384bb5f8cbe690616d", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.28, 0.35, 0.37], "teacher_probs": [0.28, 0.35, 0.37], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.6796875, -1.25390625, -1.609375], "student_probs": [0.2774980068206787, 0.42479005455970764, 0.29771190881729126], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.28, 0.35, 0.37], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "aa1cf23c63098c5607f8762bb5ae574e450647d3a0af9b646aa5aeef1e15e4ea:action", "state_id": "8f747f553a4ef9e1f4007d3022b6cd321ec274c2c0b272e66a8712aba36a05f4", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.28, 0.72], "teacher_probs": [0.28, 0.72], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.9912109375, -1.43359375], "student_probs": [0.6088266372680664, 0.3911733031272888], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.28, 0.72], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e33a04ec348c4e2099d6c1e89821c4b8367ca2083cdbac25cedce4714fa6ff8b:action", "state_id": "cb148173fba58d71fbb0c0586c81192fc84320f9b95e643abf0a5e022119797c", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.32, 0.35, 0.33], "teacher_probs": [0.32, 0.35, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.36328125, -1.8046875, -0.9990234375], "student_probs": [0.32440391182899475, 0.20863434672355652, 0.46696168184280396], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.32, 0.35, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "10be4e6620a7bc36344124b1f364f0d437614d504e3258dd5455f8dab1a67716:action", "state_id": "87089330d0b8c74b35a30ed17c06a9e3f96045d1c0e36316c93e6a6cd5fd03e1", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.26, 0.74], "teacher_probs": [0.26, 0.74], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.81640625, -1.09375], "student_probs": [0.32680830359458923, 0.6731916666030884], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.26, 0.74], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8b31d728b6b0ef71b0ad94e19faa3ba716c4e6ee075f7a1fd3e93e25f95c2ddd:action", "state_id": "06c8634521cd9fb7614f67d55d864209978855ed1c7c1476f60376d22a95e63f", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.46, 0.18, 0.36], "teacher_probs": [0.46, 0.18, 0.36], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.056640625, -1.6171875, -1.111328125], "student_probs": [0.3971914052963257, 0.2267552763223648, 0.3760532736778259], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.46, 0.18, 0.36], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "65a797f1cc9702e83fbf9a92fab8f9652325d86f23e34b594bbf143a0587b11b:action", "state_id": "22aa00ef2f7e4885ceb23750a75fafb23471cfd4943744a6e3e7659d62842746", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.52, 0.21, 0.27], "teacher_probs": [0.52, 0.21, 0.27], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.59686279296875, -1.3671875, -2.48828125], "student_probs": [0.6196860074996948, 0.28682956099510193, 0.09348438680171967], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.52, 0.21, 0.27], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7deecef3aa13cd4df0ddbd0bcea712d146a008e51a35ea8a8778c11b6472356f:action", "state_id": "21308fe08508816d1b7a95efb028800a6b4ddbbac30954e0ae774cd48da8a1a4", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.48, 0.52], "teacher_probs": [0.48, 0.52], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.59765625, -1.75], "student_probs": [0.5380124449729919, 0.46198755502700806], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.48, 0.52], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "26e92fe2138942b95cd001243e4361f6ba4d64122198c18a908aeea891cc9145:action", "state_id": "7fb617d96552dd45cd337455a8354c537e1e6b801650624f66717f79598a449d", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.44, 0.56], "teacher_probs": [0.44, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.74609375, -1.765625], "student_probs": [0.5048826336860657, 0.49511730670928955], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.44, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "70fe06a36a5c661a4c8561908849a4fdcc20d09e0932fd8f40e697ff26469900:action", "state_id": "10ba957543b55e1c98e2b73077874303c225ac744682d2a171517ef5dbaca175", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.49, 0.51], "teacher_probs": [0.49, 0.51], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.51953125, -1.375], "student_probs": [0.4639299511909485, 0.5360700488090515], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.51], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "94f0d4da0bb69bb9bd14287b982c8719e3ad0ef3790bd1bc401f6a3dd2a31f1a:action", "state_id": "0671e1085febe9125755c7eaa8068d811b60b0a9f5f917483bd288846529bc6f", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.15, 0.57, 0.28], "teacher_probs": [0.15, 0.57, 0.28], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.57421875, -0.651123046875, -1.125], "student_probs": [0.1966893970966339, 0.4950810670852661, 0.3082294762134552], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.15, 0.57, 0.28], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2ef7621669282650c436dafaee27790195ac3eab5685feb2e5e15d8c6a28e029:action", "state_id": "82dab912401bd144694482a261e0c9cf7c07bcbc4cbca186fd1135f68170ca5c", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.84, 0.11, 0.05], "teacher_probs": [0.84, 0.11, 0.05], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.255859375, -1.5390625, -1.119140625], "student_probs": [0.5886078476905823, 0.16313156485557556, 0.24826057255268097], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.84, 0.11, 0.05], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "11c6db4a0215911e77eb007c77dcb719aa4ed470b1352aee968030dfb9bf6a71:action", "state_id": "43b9a220f3a41c6441cc3feb0cd350c533559b83116ff475a7f50886d9c9e988", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.71, 0.1, 0.19], "teacher_probs": [0.71, 0.1, 0.19], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.01953125, -1.53515625, -0.4072265625], "student_probs": [0.5267899036407471, 0.11572038382291794, 0.357489675283432], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.71, 0.1, 0.19], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "934e8ef493f365cfd5f1426ddcfd0967cd4f7c2bad9969c00babf2be7b6f19b0:action", "state_id": "35c18c1e1b7f25a355979eed1764d56d872810c879df7c5c1c37eeec6c34c952", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.15, 0.85], "teacher_probs": [0.15, 0.85], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.28515625, -0.208984375], "student_probs": [0.25423112511634827, 0.7457688450813293], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.15, 0.85], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a9aaaeabe7c7a3d1532cbea3536dbec4e3d4f7ac5adafb1e69dfbde27e77abf7:action", "state_id": "0bcb7aa5598cb7c340c563c93fd4892599939c62237892f6f9cb716af9f99b1f", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.31, 0.38, 0.31], "teacher_probs": [0.31, 0.38, 0.31], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.0205078125, -1.4765625, -1.4921875], "student_probs": [0.44292229413986206, 0.2807149291038513, 0.2763628363609314], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.31, 0.38, 0.31], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8a3b4c18d649905bb5be690a5cafc011c885231f29434a1c61604ed41972b769:action", "state_id": "3e2ea6626eab81924bc42c1f62e275f5afcf19b3cd89c0371c8ea3715beac645", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.38, 0.28, 0.34], "teacher_probs": [0.38, 0.28, 0.34], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.6015625, -2.1484375, -1.203125], "student_probs": [0.3259185254573822, 0.18862716853618622, 0.4854542016983032], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.38, 0.28, 0.34], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "af4a3ecc0ef3094f0070f368fc6028df2ac583fafb0af88c1ff7bb64f7216637:action", "state_id": "f64750ca5e655479fe472b55163907685b980158b7c8791ea0906d0cc7ca56ab", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.21, 0.51, 0.03, 0.25], "teacher_probs": [0.21, 0.51, 0.03, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.37890625, 0.02734375, -0.660888671875, -0.970703125], "student_probs": [0.26255008578300476, 0.39413437247276306, 0.19803810119628906, 0.14527739584445953], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.21, 0.51, 0.03, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2754f3868b6623b43e6a7b9b19f058d5f1566ef6b2f6fae00c2ea0c68645fbfa:action", "state_id": "e641b006f810d346812ac49eda05448d24431150b4aa6ae8d6426fec0c34b80d", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.44, 0.34, 0.22], "teacher_probs": [0.44, 0.34, 0.22], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.041015625, -1.84375, -2.43359375], "student_probs": [0.5894363522529602, 0.26412761211395264, 0.14643602073192596], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.44, 0.34, 0.22], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0c7fa2f2ede41a43a8c0b607ca73352a4acc7b626aeeaa795370f59b7da7550a:action", "state_id": "f5982d9339c7958df74d834a40385c29357985f36b225bda34dde3f2c3f3c2b1", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.59, 0.41], "teacher_probs": [0.59, 0.41], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.36328125, -1.224609375], "student_probs": [0.4653874337673187, 0.5346124768257141], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.59, 0.41], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "96846365e026d62bad9818f4e68e7b6620b4e6979ef2412c6fc3825bd2d038ae:action", "state_id": "8d0c32af2a30d51d727823e95c276f52919f6f1bfbd595914dcf2e926af4fd43", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "west"], "gold_index": null, "teacher_raw_probs": [0.6, 0.4], "teacher_probs": [0.6, 0.4], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8564453125, -1.58984375], "student_probs": [0.675550639629364, 0.32444941997528076], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.6, 0.4], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "da931fc8cce635ed3b9abbaf4156fcc3a391ab1915f72641896def38883c3316:action", "state_id": "62b2eb057d060c7763a1dcb561a1b864e802e34f1e9d6c43e42476d5c913a464", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.29, 0.24, 0.47], "teacher_probs": [0.29, 0.24, 0.47], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.30859375, -0.775390625, -0.75341796875], "student_probs": [0.22488941252231598, 0.3832976222038269, 0.39181292057037354], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.29, 0.24, 0.47], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "42203735d217ff8bb1f48dcfea6e0d56753e3574950034030a8561c3451f9d5d:action", "state_id": "15d828d295a4c9c34930fa1319f08c3cbc985048aa45de3b835f2925b73b9fcc", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.44, 0.56], "teacher_probs": [0.44, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.271484375, -0.8310546875], "student_probs": [0.3916385769844055, 0.6083614230155945], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.44, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "994e6b7d2a942b5a6aefb3dfb5fbc660b82224bb9d9fa95ed8d9942f3dcaad88:action", "state_id": "40c01fc4a4ad32e6b95351cd91f414b9bc0ee48f5667fcbea4761863570731b4", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "west"], "gold_index": null, "teacher_raw_probs": [0.37, 0.63], "teacher_probs": [0.37, 0.63], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.9248046875, -0.693603515625], "student_probs": [0.44245579838752747, 0.5575441718101501], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.37, 0.63], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "38d95982bef9eb22cf06f1256b8dd999c23ea3122a4275cac63a45a0e44d7e8b:action", "state_id": "d801638458a28642ba27adc3eff3ad1749693e3aeef72f7b12d25c41c5070e2b", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.53, 0.47], "teacher_probs": [0.53, 0.47], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.208984375, -1.126953125], "student_probs": [0.7146279811859131, 0.28537192940711975], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.53, 0.47], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ca2271d85845e9096b782e12f114a9c9917417946b969f871b683392c2f5965f:action", "state_id": "b041185faa04ab614b5f7760399e61fe745598d101fd44ecaa5cb4fa529fde95", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.58, 0.3, 0.12], "teacher_probs": [0.58, 0.3, 0.12], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.16015625, -1.421875, -2.2578125], "student_probs": [0.4754253327846527, 0.3659479320049286, 0.15862669050693512], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.58, 0.3, 0.12], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "79f85bacaa4acd3498c7f8dbd53cd5e6ee19d35d0cd5656524980bd109137247:action", "state_id": "a6ff56e5c66fd0837328926db69a49e402527b7c8be2b495c49baac5d59b6fc9", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.36, 0.2, 0.44], "teacher_probs": [0.36, 0.2, 0.44], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.3984375, -1.26171875, -1.00390625], "student_probs": [0.2754673361778259, 0.31582486629486084, 0.40870773792266846], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.36, 0.2, 0.44], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4e41651da8bd390d227122e10dbb9ed8d8106a0728e366aba9407551eae808d8:action", "state_id": "265655812a1e0a6a66f83043251f47d583e17d99f3c4158b58a6d7d71f10e53f", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.17, 0.44, 0.39], "teacher_probs": [0.17, 0.44, 0.39], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.67578125, -1.263671875, -0.888671875], "student_probs": [0.2124479115009308, 0.32079628109931946, 0.46675580739974976], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.17, 0.44, 0.39], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0206c6a7e6845b5c337e80cb426158b41f3bf5b91dceeab0b1fc6de41de95741:action", "state_id": "9a68d784826899accb6bf0838b3fd54020470063e1ef75c47e30d970f323c3a3", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.26, 0.34, 0.4], "teacher_probs": [0.26, 0.34, 0.4], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.734375, -1.453125, -1.052734375], "student_probs": [0.23245522379875183, 0.3079531490802765, 0.4595916271209717], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.26, 0.34, 0.4], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7fe4ed48f6edc9c010df8459323b5280942674f2193db9c481e8027a4d2dc36e:action", "state_id": "7f193b32435c5b2d0a80dd78d53675954e9f764c347642b767d17c05f9f8720a", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.29, 0.27, 0.44], "teacher_probs": [0.29, 0.27, 0.44], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.46484375, -1.0546875, -1.2890625], "student_probs": [0.27032649517059326, 0.4073964059352875, 0.3222770690917969], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.29, 0.27, 0.44], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5a13ad14fd5b8a0e7b1c17eba4831293624dd582e3f4aaa4b00c6d2958efcef5:action", "state_id": "a681dcb81c0786a1970d8b73a3f7d481869c180fabb4bdf951fa576d5b2f5400", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.53, 0.47], "teacher_probs": [0.53, 0.47], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.3828125, -0.8466796875], "student_probs": [0.36908766627311707, 0.6309123635292053], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.53, 0.47], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9ccf54398cb6869590821e7b947104ad37026e90dffd5e0c703043f24bff564b:action", "state_id": "04c063db19d6056e19bd5f2937cd70940d233b43ebabb2f3f2e80887b3ea5e57", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.34, 0.66], "teacher_probs": [0.34, 0.66], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.169921875, -1.263671875], "student_probs": [0.5234203934669495, 0.4765796959400177], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.34, 0.66], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1f4c3de92064f779687cea30b1a514823a956f3d6461096b107e2f5d69f78e13:action", "state_id": "b1282108e68db061e3d17f71eaa37399ce35d9bab8d0b89de46c9bf1afde75d3", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.35, 0.65], "teacher_probs": [0.35, 0.65], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.01171875, -1.0625], "student_probs": [0.27904197573661804, 0.7209580540657043], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.35, 0.65], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "915fb6e44ea10c3cf4afcac49a5c788326c7c5896ae2e5af8edd416229c9c5ca:action", "state_id": "d3abcff72ec61d845cf5b453854ef2857ce3b8e16a5aafd671b29f4e55b23a25", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.11, 0.41, 0.48], "teacher_probs": [0.11, 0.41, 0.48], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.796875, -1.6328125, -1.7734375], "student_probs": [0.14315034449100494, 0.45849889516830444, 0.3983507454395294], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.11, 0.41, 0.48], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fcca450582cb083292452839b0993e09d2492d3820790ad5458afb54a32e9e1d:action", "state_id": "48455e39710106b6eb46f8c28fb91e81b40bd6590fdc6e9f023c0cabbafb2bbf", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.57, 0.43], "teacher_probs": [0.57, 0.43], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.90625, -1.53515625], "student_probs": [0.4082767367362976, 0.5917232036590576], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.57, 0.43], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fa205e62397eefcf864092acd63c10d7dffa944ad044b678ec2f90f8a6d76ad2:action", "state_id": "ed15db2199a59060f6f3f15098c6468a791f103a0b50af375e6db3f5b8f290c9", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.34, 0.42, 0.24], "teacher_probs": [0.34, 0.42, 0.24], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.14453125, -1.984375, -2.0546875], "student_probs": [0.306025892496109, 0.3591808080673218, 0.3347933292388916], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.34, 0.42, 0.24], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "031c5789256d5bcba93172e8bfe4310041f16e2574b2dd073651de4ac0a67c62:action", "state_id": "d79e17fa91f9f2534f1c73298199f786f013a3e85daa479a49b0bcb88a76dcc6", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.21, 0.17, 0.62], "teacher_probs": [0.21, 0.17, 0.62], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.4453125, -2.15234375, -1.67578125], "student_probs": [0.43720296025276184, 0.21558737754821777, 0.3472096621990204], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.21, 0.17, 0.62], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f9f0481c4fb6f867deacf3e8258c49de929036fe9242a2f50f99a71aef061e4c:action", "state_id": "4bf7a53726864219c3f0e40807e22684d506b546db923efff78083b1b74b0385", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.21, 0.1, 0.69], "teacher_probs": [0.21, 0.1, 0.69], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.36328125, -2.015625, -1.453125], "student_probs": [0.4106948673725128, 0.21389959752559662, 0.375405490398407], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.21, 0.1, 0.69], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "719a72cd28b49093cbcd61785736a2f83bc6f1721487cb090e3542b691d2c826:action", "state_id": "5b45e322c853a33de7f156add7a050f3de3e8355321cf6849b37857e8fedf662", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.1, 0.03, 0.87], "teacher_probs": [0.1, 0.03, 0.87], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.287109375, -0.739501953125, 0.599609375], "student_probs": [0.24610798060894012, 0.1565503627061844, 0.5973415970802307], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.1, 0.03, 0.87], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bbcc9b3146003c66401ab247fa0ccd26d96d9431b0a5ed45d6762451dbe46153:action", "state_id": "0db6d8102a3cd85c3c72d95b5348b2a1f1d631b4171ac26e78cef518d2b2bc41", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.46, 0.54], "teacher_probs": [0.46, 0.54], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.8203125, -2.2421875], "student_probs": [0.6039317846298218, 0.39606812596321106], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.46, 0.54], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "35dea751444cefcc239fb7f7a73f5db2ffdd37fdf933f02566cb0ae733f974b0:action", "state_id": "13d5877f3235c92507b6ec089b96b03c066d217d7999260e9d7939a41e595ef4", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.65, 0.35], "teacher_probs": [0.65, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.72265625, -2.2421875], "student_probs": [0.6270381808280945, 0.3729618489742279], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.65, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ab9ddcab99ad3ad146046e4da8a71ff5e2f41b0bd1f546cb2dddd6d799865634:action", "state_id": "ef12109bc3f8d193f7da9444ea3ea01e74dc31d80be2da4a3bc5c29adbe3fe88", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.35, 0.33, 0.32], "teacher_probs": [0.35, 0.33, 0.32], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.302734375, -1.37890625, -1.24609375], "student_probs": [0.3350159525871277, 0.3104448616504669, 0.354539155960083], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.35, 0.33, 0.32], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ad9b4916ede48a35ed0fc994c46438fb4ad9a65e05455322359f5d6c3c51a253:action", "state_id": "9dbcd74de5e88f84e24ac55d1555f7464f56b3019d511fdaac5a97b4b4738738", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.17, 0.83], "teacher_probs": [0.17, 0.83], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.89453125, -2.65625], "student_probs": [0.4407099485397339, 0.5592900514602661], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.17, 0.83], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a9d989b438fa63683679287b3e852dc944f40b1eb99c98811bdbc89c188cdc56:action", "state_id": "61849264820114f8bab126ca6a5baee784d8ed5b3ffa625c50fdfbcaf11e6077", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.21, 0.79], "teacher_probs": [0.21, 0.79], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.96484375, -2.51953125], "student_probs": [0.3904758393764496, 0.6095241904258728], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.21, 0.79], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1fe66097eb48ce38b0a2e9e581f8aa66e2b7a0850886497b3a47830700336d75:action", "state_id": "370e69ef0d3849a8bc8e030991888e119bebd829f451439f1ef992660f1edfdf", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.41, 0.59], "teacher_probs": [0.41, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.6171875, -2.0234375], "student_probs": [0.3557748794555664, 0.6442250609397888], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "993e9d534654f2056f08c62974acfe53e424c98f3edf6cb0fea97f3a38575e99:action", "state_id": "79396bc72bcadd49d82c23153567f49c6fb04efd957814acda855f30ffee538e", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.44, 0.56], "teacher_probs": [0.44, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.06640625, -0.6904296875], "student_probs": [0.4070976674556732, 0.5929023027420044], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.44, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b15c9e388a320fb29ebee0043d5d25f955c6a21f3de623f5e5fe5aea0432b31c:action", "state_id": "c21475b26821045d332d3da3d807208e25569aac6cb077bdef9c9fcdcfd696ee", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.31, 0.26, 0.43], "teacher_probs": [0.31, 0.26, 0.43], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.7506103515625, -1.181640625, -0.08984375], "student_probs": [0.27885326743125916, 0.18120978772640228, 0.539936900138855], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.31, 0.26, 0.43], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "99bd61f1df6943b2783cc7a8c7bca42d17ebe62848bdd0dcae5852411ade2b56:action", "state_id": "fb937815ab78e4eacc9795ac239cdf915d655d60e34a42426df8731689b0e69c", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.3, 0.25, 0.45], "teacher_probs": [0.3, 0.25, 0.45], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.7109375, -1.94921875, -1.31640625], "student_probs": [0.3056543469429016, 0.2408498078584671, 0.4534958302974701], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.3, 0.25, 0.45], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "dd629cddd4424951de8e6ee8e01be2947e103640b07b9f2775214afe78a223c7:action", "state_id": "fd04303bae043392a1e0f92fffb485a84f7fa86e0ca8414bea6ba50c649f5009", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.5, 0.5], "teacher_probs": [0.5, 0.5], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.6484375, -1.859375], "student_probs": [0.5525397062301636, 0.4474602937698364], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.5, 0.5], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a7694498d159901173a06b7933bc8bf9e6aacf185db8b705f8ccf7ac8ff54851:action", "state_id": "12e023a2e6d9f35d551b3c699628f6ac2024952f6456747539a09bdc1a37b164", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.58, 0.13, 0.29], "teacher_probs": [0.58, 0.13, 0.29], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.31640625, -2.1484375, -0.8583984375], "student_probs": [0.3315555453300476, 0.14428119361400604, 0.5241632461547852], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.58, 0.13, 0.29], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1896e281b3922a3509586d615d90f99a0c9711091b58c6760ef616736317fb5c:action", "state_id": "a1be29e6d6281e8f1c8cfb6dde38be2f160e4f086ad106919999ba3a560e3b19", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.14, 0.86], "teacher_probs": [0.14, 0.86], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.03125, -1.2109375], "student_probs": [0.30569732189178467, 0.6943026781082153], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.14, 0.86], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2bb233eca257c21b09ff4544ce262e1dc8dcb8bc4aa4e1c161dd17c46b310033:action", "state_id": "e27dbf93716b06577ccf3ceb38d4e6a7cdec248f3e94a2286af361c88a5b7a8b", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.59, 0.11, 0.3], "teacher_probs": [0.59, 0.11, 0.3], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.64453125, -2.05859375, -2.265625], "student_probs": [0.45489364862442017, 0.30066636204719543, 0.2444400042295456], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.59, 0.11, 0.3], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "33dfbc07ed3cf3b8ac1e34fe5d0d53e99df48f7221a7510a2cfcdf20a0019b5e:action", "state_id": "8e18b665505ac3912299cc032fa6c43231bd18cb263bf1b289a2bd8d566e8dfe", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.28, 0.72], "teacher_probs": [0.28, 0.72], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.744873046875, -1.8046875], "student_probs": [0.7426550984382629, 0.25734490156173706], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.28, 0.72], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3fd326344e27e213576a72bf3062d36ccce3a83bae169237e21e2243b3b1d031:action", "state_id": "e7609cb9e3c74e836cbf50726bb2473e24f9f4b3664878cb000ac076908fc154", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.23, 0.41, 0.36], "teacher_probs": [0.23, 0.41, 0.36], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.599609375, -0.419921875, -0.832763671875], "student_probs": [0.3345740735530853, 0.4004327356815338, 0.26499316096305847], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.23, 0.41, 0.36], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2d663ece3e2d950c4217242ee00f83bb1a122ed2e444cd79b3ee721bef8a093e:action", "state_id": "d1ff6d0b3e4027e49e52ec2ed0a198f9a76fff4599b2305a03f97cbfd0ceeae2", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.38, 0.34, 0.28], "teacher_probs": [0.38, 0.34, 0.28], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.6927490234375, -1.453125, -1.224609375], "student_probs": [0.4866175949573517, 0.2274891585111618, 0.28589317202568054], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.38, 0.34, 0.28], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4c8011d59688c3f93632b5d8b588119555286b287ce34d60fb58178e5fd94fe0:action", "state_id": "e6318c22c4063eb49287050c40a5cfe45175e3ab1ed87968a48eda480d92adcd", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.4, 0.53, 0.07], "teacher_probs": [0.4, 0.53, 0.07], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.3935546875, -0.08984375, -1.58203125], "student_probs": [0.3760016858577728, 0.5094361901283264, 0.11456210911273956], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.4, 0.53, 0.07], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "55c99dd0fd0485cb9a7bae2fc97a6c2f216e2b81d430c5c96cd2c4979f68b114:action", "state_id": "b9422b3b8babbd142ad12f4b9858d89a2de1c56ff8cd07245023f0455273570f", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.63, 0.28, 0.09], "teacher_probs": [0.63, 0.28, 0.09], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.3916015625, -0.8271484375, -2.0546875], "student_probs": [0.5445247292518616, 0.35225892066955566, 0.10321636497974396], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.63, 0.28, 0.09], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1917c886a743aa35dccac0acec58aa7e9753eabe6a7a80fe77ef4799f79488ee:action", "state_id": "b71f4e49c723874f7a18418bdb42d50507422f2a45cabafb05900897b5b4b000", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.73, 0.27], "teacher_probs": [0.73, 0.27], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.373046875, -1.2109375], "student_probs": [0.6980207562446594, 0.3019792437553406], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.73, 0.27], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0188400c069b7ec3e62ef4102e9aa5093a48b4ef2bfec47a1ba1ddeed25df7a0:action", "state_id": "8c5f1fa11238617a31e71808f2a4fcb6a1d339f64d8c084a2dbe143e26f71388", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.7, 0.3], "teacher_probs": [0.7, 0.3], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.265625, -1.6875], "student_probs": [0.8757869601249695, 0.12421300262212753], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.7, 0.3], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2bda576c52f1d4b72db567565341944ad6596bf8d4c177897eb596dfc18d818a:action", "state_id": "741b500400f27094a5a62d7908137fffeb73e74d341a988ac402d0a19bb90221", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.33, 0.3, 0.37], "teacher_probs": [0.33, 0.3, 0.37], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.9814453125, -1.578125, -1.31640625], "student_probs": [0.4413056969642639, 0.24299919605255127, 0.3156951069831848], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.33, 0.3, 0.37], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "985ea584ca10ca65565c18f86017cc6c236083babd45b4521059b77892183623:action", "state_id": "193e69d1f94d274f7dc87a1161fdfe56f8d59fe210e3b4e616ef39796e9f21fa", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.35, 0.33, 0.32], "teacher_probs": [0.35, 0.33, 0.32], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.181640625, -1.87890625, -1.048828125], "student_probs": [0.37879061698913574, 0.18861688673496246, 0.4325924515724182], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.35, 0.33, 0.32], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d963dd5fb4ae16e53560c5be23b333c1c6f28b19f6cbca6e58f048ceac5f4538:action", "state_id": "658672fadf48577346380a96eb30aac9063dbf651f057f2e00f860740bd46e8c", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.19, 0.33, 0.48], "teacher_probs": [0.19, 0.33, 0.48], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.287109375, -0.6142578125, -0.6939697265625], "student_probs": [0.20966649055480957, 0.41090816259384155, 0.3794253468513489], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.19, 0.33, 0.48], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f113d166a6193dccb474c9e98b29796cea8a457b939f539ad81051589a6241cf:action", "state_id": "627a04d670848104bd0ff47811c5444b620bcd353836b8da925335e2a7c51e9a", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.37, 0.34, 0.29], "teacher_probs": [0.37, 0.34, 0.29], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.49609375, -1.3828125, -1.21875], "student_probs": [0.29073426127433777, 0.32560694217681885, 0.3836587369441986], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.37, 0.34, 0.29], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2fd2abe6d443665728321d5fa542df7496cbd40572a1d8b943f55a5ccc868c21:action", "state_id": "4efc08a1a42b4ec2d5e982b1d15d598300e4f929dba9677b6ca08f60431da54a", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.25, 0.75], "teacher_probs": [0.25, 0.75], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.1953125, -1.80859375], "student_probs": [0.40450742840766907, 0.5954925417900085], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.25, 0.75], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fc9bb22ee5777d3f519ecba09ed407867986a3e53a48bb400e59d5e4507f0df3:action", "state_id": "f3a217e7784819f466eaa2244e46faa18bada10a760ce6f4ccdfd800acdc256d", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.44, 0.56], "teacher_probs": [0.44, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.61328125, -2.09765625], "student_probs": [0.37387582659721375, 0.6261242032051086], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.44, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "812a0ec55acc28a938b87f8822d9cb6f0987b599b4160e50184b51a09afec4a4:action", "state_id": "4f46f015f4a1951bd0d30877d646d7d4671531b92d03b8ddb00e21c1ddd69dad", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.42, 0.58], "teacher_probs": [0.42, 0.58], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.25, -1.84375], "student_probs": [0.39981159567832947, 0.6001883149147034], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.42, 0.58], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a8244b930e93ab3467c78a3d5173857cf587ce0943f63a0594722d7400388e82:action", "state_id": "ac01f6a1412808ff747e93b7ae4a7b1209d60122cf46c1c5d32d829510589d78", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.3, 0.7], "teacher_probs": [0.3, 0.7], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.48046875, -1.91015625], "student_probs": [0.36116471886634827, 0.6388352513313293], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.3, 0.7], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0a3f71447d7559a54492a840dd7f293e4ece75788d90478095924820abd1a9b4:action", "state_id": "ad9101b8607b7a55b5a6d419a774011a5cea3ea4f03930c413d176634765aa13", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.16, 0.35, 0.49], "teacher_probs": [0.16, 0.35, 0.49], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.83984375, -1.150390625, -1.0390625], "student_probs": [0.1915743350982666, 0.3817359209060669, 0.4266897439956665], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.16, 0.35, 0.49], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4adce572823d29bd49c0ba0c91934efae19c4df95fcca9799836804b74f25ef2:action", "state_id": "782bb552cfec823903baa4c160a30585473c1665be19856ab460c7429e21e389", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.26, 0.22, 0.52], "teacher_probs": [0.26, 0.22, 0.52], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.259765625, -1.181640625, -1.0263671875], "student_probs": [0.2990303337574005, 0.3233288824558258, 0.3776407837867737], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.26, 0.22, 0.52], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b091dd419ca2bd91dc62d6ce9639597cb0ec967e8e42287627e6ebd55839e575:action", "state_id": "9136f8e8ec37db3af16494f1895417892a02384bb8bda81d09e2e92d86079d6f", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.24, 0.76], "teacher_probs": [0.24, 0.76], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.7734375, -1.001953125], "student_probs": [0.31615808606147766, 0.6838418841362], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.24, 0.76], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "68281ce3c2459cea619892821eb614e68dd310885ee8564652a54d198bd12873:action", "state_id": "42de818bedbce391f1960a21149ed92a509d94562754b1d38871b6012be5f1a8", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.16, 0.84], "teacher_probs": [0.16, 0.84], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.6796875, -1.0390625], "student_probs": [0.34510529041290283, 0.6548947095870972], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.16, 0.84], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5f268a8b4e6385c5c80359425d07de52b3999a39ac4644863516192bff80a193:action", "state_id": "52c49c5f1cebe0a4d58fab1037194bc525f73f08964de0c320aa7c984729dbe7", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.3, 0.44, 0.26], "teacher_probs": [0.3, 0.44, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.19140625, -0.65087890625, -0.99609375], "student_probs": [0.254284530878067, 0.43658414483070374, 0.309131383895874], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.3, 0.44, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9c7b6e580cbc0119e40b8784a81c0ee45810da317fcec28d49587c9f327f1e5e:action", "state_id": "a1a144f95dcd525c468edeb88be332d61464832908ddf799ba92c9dfd0c6d7c4", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.22, 0.78], "teacher_probs": [0.22, 0.78], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.67529296875, -1.103515625], "student_probs": [0.6054491996765137, 0.3945508301258087], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.22, 0.78], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7cdc8786b2a1bf140e54e85f235c7a00dcd478b4f9fcebf8cd3e425867628f3d:action", "state_id": "ee7c8f21161569d11263688e8656f9ba04b880dc30a90f09286ed7c7a53934f9", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.32, 0.38, 0.3], "teacher_probs": [0.32, 0.38, 0.3], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.998046875, -0.766357421875, -1.3203125], "student_probs": [0.33498212695121765, 0.4223214089870453, 0.24269647896289825], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.32, 0.38, 0.3], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "df3e8604d92aa20d164fedc69e2a859f9efb7d3888342a98f55fc3453addf994:action", "state_id": "10968c59112e5fd88f10a3e08ba6d27005309e3b0905f8060922d8ce2401f87b", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.56, 0.44], "teacher_probs": [0.56, 0.44], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.171875, -2.109375], "student_probs": [0.7185943722724915, 0.28140559792518616], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.56, 0.44], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1990c05a55844a8d1f3b83d78241cca622a5860a5f75569f6f72c7bf90c35035:action", "state_id": "fdde84f95e40ced9e604de4adb60734de02fbdd382ec03faeac212131e917164", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.41, 0.34, 0.25], "teacher_probs": [0.41, 0.34, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.359375, -0.9609375, -0.66015625], "student_probs": [0.4370258152484894, 0.23947039246559143, 0.32350385189056396], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.34, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "14b4f559da1787db4cc959274b4cddf76f8e471c17574287dff567a99a8fdb11:action", "state_id": "b1f1668ee7b2c229b1afe541ebf802b86f4350c2781724de95f6e095fcec6445", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.36, 0.38, 0.26], "teacher_probs": [0.36, 0.38, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.777587890625, -1.021484375, -1.263671875], "student_probs": [0.4169100522994995, 0.32667768001556396, 0.2564122676849365], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.36, 0.38, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ae0017840039344a50f0c99e203464bd6967b4967c173b21d5b39b27f9824469:action", "state_id": "08b505e76f0eb89ead2e49fb1ad274ea0725906689ed3791bf519afecf4bd05c", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.41, 0.26, 0.33], "teacher_probs": [0.41, 0.26, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.203125, -0.90625, -1.154296875], "student_probs": [0.53154057264328, 0.26313164830207825, 0.20532777905464172], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.26, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "50028802a77e7fcb6cbbce40664eb4b82911c03767c031bc6a2e67a0f9b97d15:action", "state_id": "f26b5d3e51b9b5e07f2694344ffc8edc6ae69777ede2976694f8fa6132c1e15a", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.16, 0.4, 0.44], "teacher_probs": [0.16, 0.4, 0.44], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.91015625, -1.31640625, -1.69921875], "student_probs": [0.24718204140663147, 0.44758883118629456, 0.3052290678024292], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.16, 0.4, 0.44], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "87a74ff0ba37898b3b300e9387310eec4ffe934b8ea0912e985b2d20a55ef2f7:action", "state_id": "887593034c50d0cbed0d0a8a44aac27b224b64aeae9d8d892bbf23f5e27dd6f4", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.1, 0.44, 0.46], "teacher_probs": [0.1, 0.44, 0.46], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.74609375, -1.052734375, -1.294921875], "student_probs": [0.21879082918167114, 0.43767452239990234, 0.3435346782207489], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.1, 0.44, 0.46], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c1dc5ca7028bb187a64bf908440f0d3e40751ca0b05455362e641e8bdcb81ebb:action", "state_id": "5ee07f9920f374a2070def097ff1f89edacfd2ba1fd7f3614248ae9cebadc10e", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.34, 0.66], "teacher_probs": [0.34, 0.66], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.78515625, -1.48828125], "student_probs": [0.4263215959072113, 0.5736784338951111], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.34, 0.66], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2c3b4efa06da9c249add42d28ce646063fcbc343a5769683f0527b7bd83973b1:action", "state_id": "33e04314ff9688a4bc32239eb5800d98068392460b466242681b8e65cf3b7f2b", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.22, 0.41, 0.37], "teacher_probs": [0.22, 0.41, 0.37], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.44140625, -1.078125, -0.7379150390625], "student_probs": [0.22427378594875336, 0.32251474261283875, 0.4532114565372467], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.22, 0.41, 0.37], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2f38b5c7058f0833b14bb4565ef3598f69ec388eeecca30768b93c8fef61f85f:action", "state_id": "f4cf495aef45e8ffbc55dc59181d100950546be468e8603a86609d6aa167eff9", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.49, 0.31, 0.2], "teacher_probs": [0.49, 0.31, 0.2], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.87109375, -2.171875, -2.05078125], "student_probs": [0.3882333040237427, 0.2873857021331787, 0.3243810832500458], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.31, 0.2], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2d52eafc3c5c168491f94cf956a823442ef22bd331bada0e2bbb7882f982aafd:action", "state_id": "2892d5a9714fbfca146695ddc4c3facb6fe8111504b37949b01d3d050e4afea4", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.57, 0.43], "teacher_probs": [0.57, 0.43], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.875, -2.35546875], "student_probs": [0.3729618489742279, 0.6270381808280945], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.57, 0.43], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "62e3810a45ea1adc47ddab9cda03fe02f19a08466719bf29071148052f402130:action", "state_id": "86c9f49e80861fc453c22f41c7d14e0ba1923cb213ae0b2632cb34e87173738d", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.38, 0.62], "teacher_probs": [0.38, 0.62], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.77734375, -2.56640625], "student_probs": [0.4474602937698364, 0.5525397062301636], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.38, 0.62], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b23f8f2d6b57b17a282dd08fb6a6faa9ea109756b8ff480f2ff4d197ec0c2311:action", "state_id": "d56ab64d5b2cb828b6d0fa7ae25c3eb4632c0d1f58c6c4b4a17804699ff5a9e2", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.2, 0.72, 0.08], "teacher_probs": [0.2, 0.72, 0.08], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.4296875, -1.197265625, -1.79296875], "student_probs": [0.3381757140159607, 0.42666003108024597, 0.23516428470611572], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.2, 0.72, 0.08], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "608fe95d60752c2cbdf487f45072aebac3bcac07c0f776f4a066a18e14d7a7a4:action", "state_id": "739365f9418695d3af13d934265aaae1356e3ff5301fd0fe605f4d6dc7d1eb76", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.48, 0.52], "teacher_probs": [0.48, 0.52], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.38671875, -1.578125], "student_probs": [0.5477060079574585, 0.4522939920425415], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.48, 0.52], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e7310d8711a5287bb08f614d7b818a4708aa1c6f7e67f19890b83f574cbc63cb:action", "state_id": "7fb2e2cdf3099e1a023a756ab47e94a3b7358be6178019cbcf5b3915300cabc1", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.38, 0.62], "teacher_probs": [0.38, 0.62], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.60546875, -1.9765625], "student_probs": [0.34775856137275696, 0.6522414088249207], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.38, 0.62], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1647bb9d8ac1be6329c7010fa4deac56bfc0e62f1141e58a09e6af4884c417fd:action", "state_id": "98b4586a11382628f88d0eb6f63fd741464398fe2027bc0dae00ad4cfc772024", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.32, 0.42, 0.26], "teacher_probs": [0.32, 0.42, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.47265625, -2.109375, -0.970703125], "student_probs": [0.3143694996833801, 0.16630947589874268, 0.5193210244178772], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.32, 0.42, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3ad986f8e0d83d9a1e34facb38553f22b441f47b244c2797cf3bb702b483311f:action", "state_id": "6338b8606a92b25184238f6ee0bb8fa9f1acf7a0aa07c0c92e6e81ada437027a", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.29, 0.11, 0.6], "teacher_probs": [0.29, 0.11, 0.6], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.1171875, -2.14453125, -1.67578125], "student_probs": [0.28345322608947754, 0.27580755949020386, 0.4407392144203186], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.29, 0.11, 0.6], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7192076dc339f050a00aacdd8fd1b4072795b79260c940edec6d0bd5305ba835:action", "state_id": "17fef07194136b607915571e067f02bd8051cfc959a14c969b631e4d48b15463", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "west"], "gold_index": null, "teacher_raw_probs": [0.47, 0.53], "teacher_probs": [0.47, 0.53], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.3359375, -1.3125], "student_probs": [0.2643583416938782, 0.735641598701477], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.47, 0.53], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a09da7cf854455c36be87f154aa5ef4859ba135158521f3d6684bc419657a40e:action", "state_id": "884e9ac7c702b331bd8ea232f097b2aa5f1953580cace4304f615b7c1b5155d1", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.53, 0.47], "teacher_probs": [0.53, 0.47], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.3203125, -1.78125], "student_probs": [0.6132365465164185, 0.38676345348358154], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.53, 0.47], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9696899b07e9fefd98b9bbaa15bbaa71616260f12f1a6af05d52b07f6732dfc3:action", "state_id": "88ad02d10da9146b0322902a8e3c94e449129a173666c52cdbaf3a657549a13b", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.23, 0.77], "teacher_probs": [0.23, 0.77], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.19140625, -2.0625], "student_probs": [0.4678179621696472, 0.532181978225708], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.23, 0.77], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b44a378db6258ef977d1067d0f4a905150c3f23ac93a5f4d73811102232d158c:action", "state_id": "cc260e701503dcc765b330e14a121cc2cad5395186d63cb418ca079b9cf0fd67", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.1, 0.28, 0.62], "teacher_probs": [0.1, 0.28, 0.62], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.578125, -2.328125, -2.1328125], "student_probs": [0.2600778043270111, 0.3339465260505676, 0.40597572922706604], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.1, 0.28, 0.62], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bdeefb408e315c6c5726cc2a7deb2572bd40ad25bc3aa362fbf3b9d8daa390c7:action", "state_id": "12eca87eb094bb3ce1c53d11870d13ed1e6ea3742e5fe5f4d01c1f9b165c4b8e", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.11, 0.31, 0.58], "teacher_probs": [0.11, 0.31, 0.58], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.28515625, -2.83203125, -1.70703125], "student_probs": [0.13478755950927734, 0.21205060184001923, 0.6531618237495422], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.11, 0.31, 0.58], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6a7ad4b8b6e43b1bea1d4a34a12f48a471536db2238958fd57556bb7f00cdff9:action", "state_id": "52203e7c8f75270108042fa8b2f546f7521facc0af3efd5196e2b431670e9fcb", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.32, 0.68], "teacher_probs": [0.32, 0.68], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.86328125, -2.32421875], "student_probs": [0.36840569972991943, 0.6315943002700806], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.32, 0.68], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "87f844d9a3f1c5d32fffcde4b63814b5371b43294bef16ce5665846be1639409:action", "state_id": "98404d2721b25f5b46a6d366c061411c5ba527ea6617343e74f5079507007c6b", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.17, 0.01, 0.82], "teacher_probs": [0.17, 0.01, 0.82], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.498046875, -0.046875, 3.5517578125], "student_probs": [0.25337275862693787, 0.019884485751390457, 0.7267428040504456], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.17, 0.01, 0.82], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "60dd7c5c7d5cb28f298e9310a4645d1fa154b382246830d216361b09dbb6cce3:action", "state_id": "79a1712a3b3fea7c3c6f8a445f9fa30fefc7a455b789b2e849adc7f9edb54d23", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.08, 0.64, 0.28], "teacher_probs": [0.08, 0.64, 0.28], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.205078125, -0.01953125, -0.29296875], "student_probs": [0.1478842794895172, 0.48394775390625, 0.36816802620887756], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.08, 0.64, 0.28], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f786041d6b4fecd4500b94e2237e25f9bcb2326d865dc5c18658ca66b0e119fa:action", "state_id": "81f71fe0c537d27afeffb742914fb4ec9fa009ec75851f16fa6d2ebe5dac42ee", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.65, 0.29], "teacher_probs": [0.06, 0.65, 0.29], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.390625, -0.365234375, -0.02734375], "student_probs": [0.12991765141487122, 0.3622343838214874, 0.5078479647636414], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.65, 0.29], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1a95a0e3c29c968181a615a8d10553d6587a125a3a2b287b57361ff9f7f17ca0:action", "state_id": "e2c4a95b68954d8a7ee0da862b9d14f6af890bc6acfd515689d02bc674d2a709", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.63, 0.28], "teacher_probs": [0.09, 0.63, 0.28], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.90576171875, -0.998046875, 0.12890625], "student_probs": [0.21159468591213226, 0.1929415762424469, 0.595463752746582], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.09, 0.63, 0.28], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "91e4c4dbc7c26a8d002c246d0a93c132ddaec1a6d813d7bbc7e31f3e74e51294:action", "state_id": "d940347f89d5a65d232962f54e9dbc910b70252bc2cfeaca7d16f683c48ae1fd", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.26, 0.2, 0.54], "teacher_probs": [0.26, 0.2, 0.54], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.35546875, -0.7841796875, -0.744537353515625], "student_probs": [0.216793030500412, 0.38384243845939636, 0.39936450123786926], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.26, 0.2, 0.54], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "675356819ecf3fe902a7c8e4a12ab1e00a41ba6808e202ec32c68875fd435366:action", "state_id": "c074a8f66f8df854b32748561a76c95b4add2ec612f24a73919ba519b26a33f4", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.43, 0.02, 0.55], "teacher_probs": [0.43, 0.02, 0.55], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.126953125, -2.13671875, 0.34375], "student_probs": [0.36561205983161926, 0.0489993542432785, 0.5853886008262634], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.43, 0.02, 0.55], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "16913b1e809dff5e7accabdd3f05fae37346680dee9a66d8ddafcfb401985eef:action", "state_id": "22e6ea370b14b5a320a29f7f09950132ec6dd23ed5d4ed2234c4587e0813bc27", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.48, 0.02, 0.5], "teacher_probs": [0.48, 0.02, 0.5], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.46875, -1.875, -0.05078125], "student_probs": [0.5914477705955505, 0.05675951763987541, 0.35179272294044495], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.48, 0.02, 0.5], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4d75ff11389c93d0200579ff5987a546bd381b67d25b61ec3b51d4becf904a56:action", "state_id": "c1fe20ff47d2e98018ed566840ff962ed5b857d81284eb768dce2b9d6b5d3a18", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.62, 0.02, 0.36], "teacher_probs": [0.62, 0.02, 0.36], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.00390625, -2.30859375, -1.05078125], "student_probs": [0.6892639398574829, 0.06878162920475006, 0.24195441603660583], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.62, 0.02, 0.36], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "44b50fd612080fb015d1773fd1cda99948e1701f9eca8a475c6e454e51d11809:action", "state_id": "af365708fd8cade91de4882a7e2494dea043892b109787c9e2932f3843b34b97", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.08, 0.8, 0.12], "teacher_probs": [0.08, 0.8, 0.12], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.017578125, 0.30078125, -1.82421875], "student_probs": [0.19291463494300842, 0.7209769487380981, 0.08610841631889343], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.08, 0.8, 0.12], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "96ace41fd5283c0c3a7f34e2ff3086950777c6f381ec3df78d92b895edfa0ff1:action", "state_id": "3309bab928a3ba11c967ed98ff2ac0b840a1308c0d8a5ae8905da954e4fcf994", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.96, 0.02], "teacher_probs": [0.02, 0.96, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.53515625, 0.796875, -1.3828125], "student_probs": [0.1916755586862564, 0.7262072563171387, 0.0821172371506691], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.96, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fb7e40f9a5afe30037979ce47f7082b1777a9b7b9f82fb4aa7ead21aa99a9f42:action", "state_id": "fa9c03f5dfd4dd1c1d2bfa0557b7e82ec59efb789b60cd45f0d69cc55912b669", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.99, 0.0], "teacher_probs": [0.01, 0.99, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.11328125, 1.607421875, -0.7943115234375], "student_probs": [0.14095324277877808, 0.7877110242843628, 0.07133577018976212], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.99, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "512a4b84af38b353b83d2bc595ebcd1c4ad38ec21e7b90c2ac4878508fc42d02:action", "state_id": "d47319bc49403c76a54232ced02254f48acfd9ad6eabbba90b95d6d5da270546", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.55, 0.13, 0.32], "teacher_probs": [0.55, 0.13, 0.32], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.015625, -1.173828125, 0.24609375], "student_probs": [0.6348367929458618, 0.07108774036169052, 0.29407554864883423], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.55, 0.13, 0.32], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "492b58b822fff2c8e176f6d8f13e67f4a244935604ba60dc72c249b6417b0695:action", "state_id": "d734779954ee848b9daf5698f9b11bb41546c00fd44a78d00265df41f47592cb", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.56, 0.43, 0.01], "teacher_probs": [0.56, 0.43, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.5625, 1.896484375, -1.359375], "student_probs": [0.20232796669006348, 0.7680649757385254, 0.029607122763991356], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.56, 0.43, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0de206762f6d9ca032f05791f8a8a84ec7a297b885411b58508cb24e02da4d96:action", "state_id": "bff7e864cdbfebd3832d72f19c0e3e88db6c1c35f5313f7e0f2bfc4b2f1f79b8", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.09, 0.9, 0.01], "teacher_probs": [0.09, 0.9, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.505859375, 3.796875, -0.730712890625], "student_probs": [0.09097695350646973, 0.8993045687675476, 0.009718525223433971], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.09, 0.9, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1d52188a5e62ed1494bc99b506b5e6bdeb7c639355e193144f61a7e65d29d3b1:action", "state_id": "2dd830ebb05ac47963c4a24acb0e36f2e162d48dca7d6ddd8da16a3c7b647b1e", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.53, 0.39, 0.08], "teacher_probs": [0.53, 0.39, 0.08], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.02734375, 0.32421875, -1.51171875], "student_probs": [0.6353335976600647, 0.3145129382610321, 0.0501534678041935], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.53, 0.39, 0.08], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "48481b57e5f108169ec390d4228e0c96608da026dee31d9b077d72f31f331da1:action", "state_id": "f8fd26f49b7c93416a402f60d825215076ef90180c050e3611704968413f483c", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.48, 0.17, 0.35], "teacher_probs": [0.48, 0.17, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.248046875, 0.08203125, 0.3359375], "student_probs": [0.5836750268936157, 0.18187665939331055, 0.23444828391075134], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.48, 0.17, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8b62574d44e29213d8fb7a7b4ccdb92f9d09a144b150fc1ea637c79c60dab79c:action", "state_id": "9223f5287e2e8f14336aca70fc54d11e1df4c19469f2ccd3693dd607f660e6dd", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.55, 0.23, 0.22], "teacher_probs": [0.55, 0.23, 0.22], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.814453125, -0.294921875, -0.5419921875], "student_probs": [0.6299849152565002, 0.2077469676733017, 0.16226819157600403], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.55, 0.23, 0.22], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bdb302ca2243f86557c479c561b6c3316fa7d18ee9480225161f763cecaf597b:action", "state_id": "6b0af43678397fb64b4d0bfbe1fc21623cb4e66842ae4deb278c3e1a8b36701d", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.52, 0.23, 0.25], "teacher_probs": [0.52, 0.23, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.01953125, -0.228515625, -0.66455078125], "student_probs": [0.6790342926979065, 0.194926917552948, 0.12603877484798431], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.52, 0.23, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "421b122bb8c6e0e8d5caa844b1a0fa0d96e006facb8298d974aadab905761bc3:action", "state_id": "ca13bdf3165a54ae807c7baf3bdcc246070ae71a77289911821713aed93674aa", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.49, 0.23, 0.28], "teacher_probs": [0.49, 0.23, 0.28], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.60546875, -0.12890625, -0.16015625], "student_probs": [0.5141788721084595, 0.24670571088790894, 0.2391153872013092], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.23, 0.28], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9f57a1de1c9137337377b652ea0266f7e503d73c662b67b3d2711388841a16f8:action", "state_id": "25123e7f382fa9677492db07a024b0a1128c96378860fa0c73f0c34c432b051b", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.52, 0.27, 0.21], "teacher_probs": [0.52, 0.27, 0.21], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4296875, -0.388671875, -0.6728515625], "student_probs": [0.3538556396961212, 0.3686710000038147, 0.2774733901023865], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.52, 0.27, 0.21], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6f3660b03ec8d1e3a80b2091d0afb62ce2eea3e96191cebe8b287e7cf47fd6b9:action", "state_id": "d14f3b3443ac5d4244bdb992fed78b3abb312defc3e559e8009314ae3abe71af", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.44, 0.35, 0.21], "teacher_probs": [0.44, 0.35, 0.21], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.1171875, -0.785430908203125, -1.1875], "student_probs": [0.3007051944732666, 0.41900673508644104, 0.28028807044029236], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.44, 0.35, 0.21], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "970834c7bfb1b5248b1e01cc380709f76f72e27814a25939e6b110542d726f94:action", "state_id": "9777e7fa49d20f20378678e5ddbe73333a268ab97bcfbfdbdfd31c449991d2f6", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.05, 0.59, 0.36], "teacher_probs": [0.05, 0.59, 0.36], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.5625, -1.0703125, -1.63671875], "student_probs": [0.2805553078651428, 0.45895788073539734, 0.26048681139945984], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.59, 0.36], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "aea803002905de8ed9707ec225e1513db39b1de5fb8ce522274ac3b71a0b8279:action", "state_id": "47090cb20de46c97d1d1adf030dca8f837468f2405b24f55029e95c294861cab", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.74, 0.16, 0.1], "teacher_probs": [0.74, 0.16, 0.1], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.228515625, 0.06640625, -1.046875], "student_probs": [0.7064230442047119, 0.2209872305393219, 0.07258974760770798], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.74, 0.16, 0.1], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a4a6bc3f1cebfee52711813f9c62ba4a7ff6ca0eebc11c8624b6a6b5bf638f01:action", "state_id": "757edb06203244564fba5c4a4ae1d158173c8cdfab9f1a190452ab736e0642cd", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.63, 0.21, 0.16], "teacher_probs": [0.63, 0.21, 0.16], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.733154296875, -0.17578125, -1.060546875], "student_probs": [0.28844377398490906, 0.5036457777023315, 0.2079104781150818], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.63, 0.21, 0.16], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f5d7d8808e20459a29e478c85d86ac046ed44d6c33d8d1ab6806a7adc6e3a6a9:action", "state_id": "dd3e85afee4d3cb1c48ee7212a02463238df35c4b45a6e32a039c1372af08ac1", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.44, 0.35, 0.21], "teacher_probs": [0.44, 0.35, 0.21], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.6953125, -0.66455078125, -1.1953125], "student_probs": [0.1834215521812439, 0.5141673684120178, 0.3024110496044159], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.44, 0.35, 0.21], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "773b31775e7942a3e30bc8d329e074a50a7ff2e5fbbcb4a343a09e1ec578973f:action", "state_id": "044107f02135d56c820350663d5af9df0dc6994debb3817e7480d478ff07b720", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.12, 0.53, 0.35], "teacher_probs": [0.12, 0.53, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.84765625, -1.42578125, -1.82421875], "student_probs": [0.2818066477775574, 0.42970383167266846, 0.28848952054977417], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.53, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c0c3c8b31e82a8adeb7a16808e9ac23efc06735f08170d05882190c3f57accfd:action", "state_id": "a2a292368f0c50702fdee02a2372116796263d43a963c73b0c755f2f380d617c", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.4, 0.59], "teacher_probs": [0.01, 0.4, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.703125, -1.24609375, -0.70263671875], "student_probs": [0.07882792502641678, 0.3384236693382263, 0.5827484130859375], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.4, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "861a7a2e4039416caaa79ecafc35022048e34529ebd596ec9b04ad3c8d7dac83:action", "state_id": "5797af05e3ea650652c9de3de976f8d2e63e764e53b47d22a3e619cdfc58eb93", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.14, 0.17, 0.69], "teacher_probs": [0.14, 0.17, 0.69], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.9833984375, -0.660888671875, 0.69140625], "student_probs": [0.12956152856349945, 0.17887113988399506, 0.6915673613548279], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.14, 0.17, 0.69], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9fb163f075354e4c806d7ec326f70221b7a4126b94a11b2e774f57e994f8feeb:action", "state_id": "f2fcc0c8d2ddf08a22c7318aa241de68532690ef504a4a904bfe19c5b51e39e2", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.13, 0.07, 0.8], "teacher_probs": [0.13, 0.07, 0.8], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.412109375, -0.16015625, 1.298828125], "student_probs": [0.12786607444286346, 0.164504274725914, 0.7076296806335449], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.13, 0.07, 0.8], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0e2747eb4ea06106c76c11fa022b324d86e61606788bb0d70569db287f191d84:action", "state_id": "c6435250a90544a70bb406a15e78abfa23d3e81872cc9849087309919bc28459", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.14, 0.1, 0.76], "teacher_probs": [0.14, 0.1, 0.76], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.015625, 0.14453125, 1.716796875], "student_probs": [0.1312689334154129, 0.14932934939861298, 0.7194017171859741], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.14, 0.1, 0.76], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4eddb756701b94c10087bd47ccbc4b90dea1e738a10ae2f087c2178a21f1a2b6:action", "state_id": "24246e851c0a8cbe526654140c2c37f0ac4f0d4772466f9880fd70664d3f6f78", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.17, 0.22, 0.61], "teacher_probs": [0.17, 0.22, 0.61], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.41796875, 0.759765625, 2.255859375], "student_probs": [0.11506493389606476, 0.16195093095302582, 0.7229840755462646], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.17, 0.22, 0.61], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "70e6455ee4cf404b2dd0875efd99882def94f6dd23ce7489c5af80e34df48474:action", "state_id": "63ab22bcfda2ec423d970ad4c9b1f98882811749727aad49a488453102b1def8", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.14, 0.19, 0.67], "teacher_probs": [0.14, 0.19, 0.67], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.60546875, 0.5859375, 2.306640625], "student_probs": [0.13402986526489258, 0.1314374953508377, 0.7345326542854309], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.14, 0.19, 0.67], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c715e86e50559ec7d67b1708bf9757a3326fab5568e63819ecdcfe031de0e205:action", "state_id": "58b4d25c18fe837664d733cca78ea7b92ab7f8d14dc7d0132ac7a14652d9e030", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.14, 0.26, 0.6], "teacher_probs": [0.14, 0.26, 0.6], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.52734375, 1.56640625, 2.046875], "student_probs": [0.11909513175487518, 0.336630254983902, 0.5442745685577393], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.14, 0.26, 0.6], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "96539ddf05b27a4abaf1b02b212efa4cb9db1b6ad7c8a49534044d530d3aef4d:action", "state_id": "19c897679e025a25b6c7534ba786e022acfd4fb58ca662de00766cdacccb4bad", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.22, 0.38, 0.4], "teacher_probs": [0.22, 0.38, 0.4], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.6796875, 0.02734375, 0.12109375], "student_probs": [0.19028618931770325, 0.3858931064605713, 0.4238206744194031], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.22, 0.38, 0.4], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d63911941430dbfae959bf1b8ebb0afcfd30bca90c070262dfd1f376a35e06bf:action", "state_id": "a3ac6babdbeae760955db69ab2a42c8dc07d63fabf5e9a7d0e23ffa9ffd65e5b", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.4, 0.53, 0.07], "teacher_probs": [0.4, 0.53, 0.07], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.53515625, -1.064453125, -2.37109375], "student_probs": [0.3295340836048126, 0.5276234745979309, 0.14284244179725647], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.4, 0.53, 0.07], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7b101a572c8d65a9306e7f59d887e2949c8014e9c769bb282e0e439d15ac70f0:action", "state_id": "682833b9f76e59de79c79137fa700fa46a98768d5250354c90da8ccfb5fd3d0c", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.25, 0.04, 0.71], "teacher_probs": [0.25, 0.04, 0.71], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.625, -0.4619140625, 3.009765625], "student_probs": [0.19539037346839905, 0.024241970852017403, 0.7803676128387451], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.25, 0.04, 0.71], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "382dae067da364013937721d9e0ae1a0cb7d49dec9d6368aa556f0ebc89e6b57:action", "state_id": "8ef3d35862eb4f0098fb5850dd6701babd86e68b16ad8f9f64930093663df793", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.51, 0.46, 0.03], "teacher_probs": [0.51, 0.46, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.109375, 0.1640625, -1.984375], "student_probs": [0.45883458852767944, 0.484625905752182, 0.056539516896009445], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.51, 0.46, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "46eb658ba3a336dcb588f76b2f14105b762b88683469deb0857234bdb45f7255:action", "state_id": "ed41d6dd7d916b999c48358c797f9fbfefcce83b6fa765b24810023c441040aa", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.72, 0.08, 0.2], "teacher_probs": [0.72, 0.08, 0.2], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.455078125, -0.28515625, 0.1328125], "student_probs": [0.6934764385223389, 0.12169073522090912, 0.18483279645442963], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.72, 0.08, 0.2], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "18feb991923be1379cfd6e86f4bf52652d61c0e2700720ab5f75b595c4ef4e87:action", "state_id": "416a0a70d9d13718eed8041109d97672c16fdf16a081727d4800f6b4b8fdd66b", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.66, 0.12, 0.22], "teacher_probs": [0.66, 0.12, 0.22], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.775390625, -0.15234375, 0.59375], "student_probs": [0.6885855197906494, 0.10017365217208862, 0.21124084293842316], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.66, 0.12, 0.22], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d58f614b125f3ddfae364111f7cd29a3af444971fcf1ae17251a06c477c2e2cd:action", "state_id": "2214b8a136cf33aa7ce34ff834f92be87334280b0450bccd6447a4a2b2083503", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.73, 0.09, 0.18], "teacher_probs": [0.73, 0.09, 0.18], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.560546875, -0.1171875, 0.37109375], "student_probs": [0.67060786485672, 0.12526734173297882, 0.20412477850914001], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.73, 0.09, 0.18], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4f55c320a0fe530f2801e8e78dbb775b8d4e03b77a2591df303a22e90a75a01c:action", "state_id": "f7949725e2672d54d845726ebbc60cc103caf2458f76db0d446c4ad1767891a7", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.52, 0.22, 0.26], "teacher_probs": [0.52, 0.22, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.974609375, -0.0234375, 0.6796875], "student_probs": [0.7094618678092957, 0.09620293229818344, 0.19433526694774628], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.52, 0.22, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e0eb866639646ff9e824a34f0fc7e6f313dd3b95b0827147b97f757f460fd7c8:action", "state_id": "91f6f8574c181c14196de718b94f266b0bba0d12ab679b710d71079f6ec1a77a", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.07, 0.25, 0.68], "teacher_probs": [0.07, 0.25, 0.68], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.2734375, -0.70050048828125, 1.86328125], "student_probs": [0.038757212460041046, 0.06873468309640884, 0.8925081491470337], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.25, 0.68], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a87aafdbe0a2badb1e09f222cc7fae2b6884f71d53a4b48d3cf5da22abdbbfb5:action", "state_id": "c9e7a3ae1786026f8340b5390fcba05710e37ed714d38a6ee25785aae2773523", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.11, 0.88], "teacher_probs": [0.01, 0.11, 0.88], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.9765625, -1.37109375, -1.22265625], "student_probs": [0.08505340665578842, 0.42358243465423584, 0.4913642108440399], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.11, 0.88], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c693132b99147b1522c06bb9af118bc6272ddc6928d9d38ad8f231e20d57e540:action", "state_id": "60137bd863c4941b28b86d5a754a93a911501f1bcb3848fdf2dbf379dcb035c1", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.53, 0.01, 0.46], "teacher_probs": [0.53, 0.01, 0.46], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.6875, -2.03515625, 0.63671875], "student_probs": [0.49598580598831177, 0.032586272805929184, 0.4714278280735016], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.53, 0.01, 0.46], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "20396ca5b5f868be76baa10769caabc38bf70477c8e3321f8419bfe521e7d2df:action", "state_id": "6554179006bf1036f6edd65128ad094021bdefe759ceeb70a1ff3887fc02d760", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.9299999999999999, 0.03], "teacher_probs": [0.04, 0.9299999999999999, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.169921875, 1.28125, -1.146484375], "student_probs": [0.07339099794626236, 0.8514775633811951, 0.07513141632080078], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.9299999999999999, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ce18a18e27e9078ab3679024d69f9393dd4e8d291a46fb0a8dc8363fbdc3d51b:action", "state_id": "c3a883c608d2bb3d5bada682554b42d38d1843e15afe073024e3eb07f40e8131", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.98, 0.01], "teacher_probs": [0.01, 0.98, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.97265625, 2.55859375, -0.79296875], "student_probs": [0.01029569935053587, 0.9562087655067444, 0.03349559009075165], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.98, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "22358c5565a21aab439fbf871f3b81fa53d3eb29da691e6ddf1773b5fc525b61:action", "state_id": "d5ac7e3e75a212a5b3dfcdb4643aa273d500f2ccaed2346aecf02141888be948", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.76, 0.22, 0.02], "teacher_probs": [0.76, 0.22, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.171875, 0.3515625, -1.4453125], "student_probs": [0.6608066558837891, 0.2909492254257202, 0.048244114965200424], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.76, 0.22, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "747c95055e19cbe652b092985ca5193ac10230fb188a702d87149162aaa268a9:action", "state_id": "a393d4dc050fd5429208560f8bc25a4c6597e330f4c9098c1b06c058890a7576", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.67, 0.31, 0.02], "teacher_probs": [0.67, 0.31, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.8984375, 0.8046875, -1.328125], "student_probs": [0.49543970823287964, 0.45110300183296204, 0.05345729738473892], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.67, 0.31, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5cfb499a3d3355cd9da0561c0181c3d02b5e559dbb52656b5cf663f1dce7020f:action", "state_id": "faa8c522a9672686bf403f5d556d144d7807c65e2652a29848d3746ba94f5514", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.56, 0.4, 0.03], "teacher_probs": null, "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": null, "student_logits": [1.265625, 0.66796875, -1.328125], "student_probs": [0.6154457330703735, 0.33855634927749634, 0.045997947454452515], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": "Rounded target has no identity probability-simplex representative", "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": null, "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d29128d4244a25448320db66f4c0fc66c231ca02a7400b92eb7af9d6c7b9fa00:action", "state_id": "0ac6a87e88042472f196171941e64dc416b3f29bf5c45dafaabb95594c147591", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.49, 0.45, 0.06], "teacher_probs": [0.49, 0.45, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.20703125, 0.390625, -1.146484375], "student_probs": [0.6505961418151855, 0.28757476806640625, 0.06182905659079552], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.45, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4f554328b57b01a6884ff73f7f935463e88e2729e0063c6c4773cf953f9ba4f7:action", "state_id": "5e82eeaab2230a44521eb5c69b23257b7d832e28ea9f344b6c677340fb1aff48", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.13, 0.79, 0.08], "teacher_probs": [0.13, 0.79, 0.08], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.41015625, 0.6875, -0.857421875], "student_probs": [0.09186911582946777, 0.7484625577926636, 0.159668430685997], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.13, 0.79, 0.08], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "61d6857cfba1402c4d18277a75e7a1c68b582d511756c59d52894aa2a411bcc3:action", "state_id": "41e3205885f721c3c74eb39aa10bb6738bf463f1a8b56c2c50f9620512538eae", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.94, 0.05], "teacher_probs": [0.01, 0.94, 0.05], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.01953125, 1.4375, -1.86328125], "student_probs": [0.01106083020567894, 0.9537878632545471, 0.03515124320983887], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.94, 0.05], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4d87f5eb6f6402e32a0d537ce9d75398a74be88f0ff2647716b7dc59f54d5fd6:action", "state_id": "88698e8f916ddb27a387fa2ff34760175b8998f2880f526571f5f81bbe673e8e", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.0, 0.96, 0.04], "teacher_probs": [0.0, 0.96, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.16015625, 2.625, -1.037109375], "student_probs": [0.00807791855186224, 0.9670888781547546, 0.024833189323544502], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.0, 0.96, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6c7fd7e2f093813b92f4cf4ab032d3efe27512e2e2b68636b99a0df5c1c1728e:action", "state_id": "cd96ce4adfcc0c07e9dd7ad70034458a394347b6d7c610059f91f2059ddb7000", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.0, 0.99, 0.01], "teacher_probs": [0.0, 0.99, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.9765625, 4.40234375, -0.7138671875], "student_probs": [0.001684018294326961, 0.9923630952835083, 0.005952897481620312], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.0, 0.99, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ec21242525c09b5e3b2f1301769aa905fec580611e84c6cfa999f092b1be6b8f:action", "state_id": "5022793fafad2ba83a1075310ae630c54ea07e0a7eb5f2c91137c53d6e75a8f1", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.71, 0.28, 0.01], "teacher_probs": [0.71, 0.28, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.646484375, 2.212890625, -0.61767578125], "student_probs": [0.5929775834083557, 0.3843535780906677, 0.022668955847620964], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.71, 0.28, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0d7013baf66ed25b35cf7ef3ec652694bf5b236cec2f49a158ba49ffb929af60:action", "state_id": "55ce20ddc053b4c63b201d81934481ae232e6b1f1e49f6cd0cd2b7e6032357a1", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.12, 0.88, 0.0], "teacher_probs": [0.12, 0.88, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.2734375, 4.025390625, -0.8310546875], "student_probs": [0.05954110249876976, 0.933200478553772, 0.007258511148393154], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.88, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2ce5d70855fd803756ba44ce798433b049c665e661434b2abaf5ca3598475021:action", "state_id": "68fb19f1ec33be7c2af62f017e4c665b39e867122673a0fb8345c293e790df4d", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.79, 0.14], "teacher_probs": [0.07, 0.79, 0.14], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.28515625, 0.76953125, 0.10546875], "student_probs": [0.07799241691827774, 0.6086839437484741, 0.31332364678382874], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.79, 0.14], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d1d06185580a2928dbb19eebbe89046e69c19d0b476cae67b8829e7f3f35f2f4:action", "state_id": "abb2ba5c7eeb9f179ee431718914a0553d20bfcd225e8958f4aa12f0eb03ffdd", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.61, 0.32], "teacher_probs": [0.07, 0.61, 0.32], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.28125, 0.64453125, 0.546875], "student_probs": [0.07100903242826462, 0.4871579110622406, 0.4418330490589142], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.61, 0.32], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "10471614b338607c5d9597627dc239cb1d0644b7766eafb1b1df0a56ec78255a:action", "state_id": "bfb5e978308398f553a85ee0108b5f8149f876a4fc10643ec28db110d1816aa1", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.31, 0.65], "teacher_probs": [0.04, 0.31, 0.65], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.060546875, -0.00390625, 1.0859375], "student_probs": [0.08044132590293884, 0.23140482604503632, 0.688153862953186], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.31, 0.65], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "dc82f4f24333eb90792f22ab1019385fa9fa1e5c130332386a58347d7c23214b:action", "state_id": "bcb2a6af304aff7a5b17ca1f0fa30dff5ab44bf849f72df9a3c89b0e7d51403b", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.03, 0.9], "teacher_probs": [0.07, 0.03, 0.9], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.01171875, -0.5166015625, 3.51953125], "student_probs": [0.02795621007680893, 0.016873706132173538, 0.9551700353622437], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.03, 0.9], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a80c96df3dc8a54cc46aaada7c8ec3b91d3516c31ce2d6b46b2d81e443a8cdd6:action", "state_id": "df7d2a0bda1d6e72c7308c94db7be255508960f3d403def1e2ca16f5a39c8ab8", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.02, 0.94], "teacher_probs": [0.04, 0.02, 0.94], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.22265625, -0.7248382568359375, 3.4111328125], "student_probs": [0.039002832025289536, 0.01512183528393507, 0.945875346660614], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.02, 0.94], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1e239fb1794a7661136f919f3e6efabcdffdb7662b921d20ff4de73cfceb9a6d:action", "state_id": "58dfc4b42247db35ba7f26fb31bac59a2a1365f8b582c3d05ededb219f834db2", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.01, 0.97], "teacher_probs": [0.02, 0.01, 0.97], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.625, 0.01171875, 4.4169921875], "student_probs": [0.021793032065033913, 0.011802473105490208, 0.9664045572280884], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.01, 0.97], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "af9a5a8bdf7db7f34a102bd908e1a322063aa649f04681dd79a67583dcaa6233:action", "state_id": "77243696784fa332bf4981c0d1cfbd55e4744f08038fac12229bd76a8c8add56", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.69, 0.06, 0.25], "teacher_probs": [0.69, 0.06, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.1796875, -0.7891845703125, 0.44921875], "student_probs": [0.6167899966239929, 0.08611266314983368, 0.2970973253250122], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.69, 0.06, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "995f4692c4e254c7173b30574e499bb6192966f0d02366ef993ecc9892c46280:action", "state_id": "02871fa2f8b5fa9f2ae87c6c025ba3aa19c53d6949e89fcb2254e3934250818a", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.95, 0.04, 0.01], "teacher_probs": [0.95, 0.04, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [3.41015625, 0.75, -1.2578125], "student_probs": [0.9265018701553345, 0.06479703634977341, 0.00870108138769865], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.95, 0.04, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0c11c4030767db58b21de2229c72ff75e04ddf7629ff6d26c2aae24c95dd968e:action", "state_id": "0440b83075b70a686f8b1f8354bd8d0909a3875359047110c02ce96c8c8e68ab", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.8, 0.18, 0.02], "teacher_probs": [0.8, 0.18, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.71875, 0.1796875, -1.64453125], "student_probs": [0.8005099892616272, 0.17177517712116241, 0.02771483175456524], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.8, 0.18, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "240876893014f9224c049691a9bb3c02ac38d41029304396ddb0b1e6ff50d2c6:action", "state_id": "0461cc88363f25cb0c1554ef634954996c9e6e61129d6b300faa70b1a1c19f03", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.72, 0.26, 0.02], "teacher_probs": [0.72, 0.26, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.4453125, 0.8359375, -1.30859375], "student_probs": [0.6221346259117126, 0.33824872970581055, 0.03961668163537979], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.72, 0.26, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1fd86fcbfe70f449a449844aa3d801132a128acd55791d84b0f2e5933007ad6d:action", "state_id": "33f8c430543970f0c31976b435a4305d95d946d4f8043211d3dc8be58b280d17", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.68, 0.3, 0.02], "teacher_probs": [0.68, 0.3, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.65625, 1.00390625, -1.01953125], "student_probs": [0.6290587186813354, 0.3276286721229553, 0.04331258684396744], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.68, 0.3, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c0c3f4439cb0f91f8819db2d370398fa7f7a3c66edcef7b2b0ce648fa5868b5e:action", "state_id": "645b7aad5e4d63defb98659aab0a7bce9ddd81a911cc3de5640a193e87260504", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.31, 0.64, 0.05], "teacher_probs": [0.31, 0.64, 0.05], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.751708984375, 1.564453125, -0.12109375], "student_probs": [0.07683169096708298, 0.7788195013999939, 0.14434877038002014], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.31, 0.64, 0.05], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "74a837e0b744ef27f09e8667b5a26be732863d86d0d23ab8e47ada1c40f4ea05:action", "state_id": "5acf9f227815afb14c185f9a86d217429581b51a908ebdf85a0dc448eb1cd9c1", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.96, 0.02], "teacher_probs": [0.02, 0.96, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.7109375, 2.3125, -1.21875], "student_probs": [0.00635406794026494, 0.9653905630111694, 0.028255347162485123], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.96, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cffb4fe61056f5ef14368edad95300cce4610d2cb9c99c6f7e999a7feae492c9:action", "state_id": "22b9611bb0b1993435297480f69d16c853ed0416375e81424f9f3e581807a645", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.0, 1.0, 0.0], "teacher_probs": [0.0, 1.0, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.40625, 4.296875, -0.5433349609375], "student_probs": [0.0012159666512161493, 0.9909502863883972, 0.007833852432668209], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.0, 1.0, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7d649f8e6e70d1d39bf4a91bb240800a97c89365d8734824c32cfc38609d9a02:action", "state_id": "b644252ef8f642911ffbbb7b623cca24db02b3f73ba7ff520f7ea64768bb8e7a", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.31, 0.64], "teacher_probs": [0.05, 0.31, 0.64], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.26953125, -1.125, 0.265625], "student_probs": [0.0228110458701849, 0.19476157426834106, 0.7824273705482483], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.31, 0.64], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9c37c72d7370afd33868565bf751a8cc8ee183e05c6f068f4a050825783f6ea7:action", "state_id": "022322c2ba4142b0422b744837495fa921a052c6326e6b1d85d0ea2b4b18dd2e", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.11, 0.19, 0.7], "teacher_probs": [0.11, 0.19, 0.7], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.24609375, 1.09375, 1.41015625], "student_probs": [0.0994226261973381, 0.37964001297950745, 0.5209373235702515], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.11, 0.19, 0.7], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a53034012773a25fa460585c2577c850f76a21652cb1f102d772f65f330ede2d:action", "state_id": "f746e879455d8d699ae25dcc3fb7e5040ddcb782b8e7280399bc186f2ae6fc13", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.08, 0.32, 0.6], "teacher_probs": [0.08, 0.32, 0.6], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.1953125, 1.115234375, 1.798828125], "student_probs": [0.08295940607786179, 0.30763015151023865, 0.6094104647636414], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.08, 0.32, 0.6], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bb9a63a665b96846a5b5b2e1af849cd9ea63edcb977f81ebbca1af6986486628:action", "state_id": "f1289aa1e46e7aeb0c402b881113f067f30a3ed1b0cd8cda1834b3906ff62e77", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.8, 0.14], "teacher_probs": [0.06, 0.8, 0.14], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.8359375, 3.32763671875, 1.232421875], "student_probs": [0.06864182651042938, 0.8293160200119019, 0.10204219818115234], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.8, 0.14], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c4ad263f041951c300898f3470e183e403a6f40f39c574f7228cc08e4bbdc00d:action", "state_id": "7d87799ca16cf5e72013470511313a7c52383c5c5b545fdab2a621471cb4c65d", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.25, 0.58, 0.17], "teacher_probs": [0.25, 0.58, 0.17], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.01171875, 0.7421875, 0.078125], "student_probs": [0.10255426168441772, 0.5924688577651978, 0.30497685074806213], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.25, 0.58, 0.17], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7fa0e29729290352f1f22131323e9b899f385f951d136e9c04ecd92a42f7417f:action", "state_id": "27beecd9c9337317303e501b2d14f7ea552523443404e13bc6cd8b7fca003276", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.2, 0.67, 0.13], "teacher_probs": [0.2, 0.67, 0.13], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8955078125, 0.640625, -0.05859375], "student_probs": [0.12569421529769897, 0.5840489864349365, 0.2902568280696869], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.2, 0.67, 0.13], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "59b6d4c451b581a51bdddba06d5001ca65ad853b83d62ec2d22f6b58be426aaf:action", "state_id": "7de45c94d718f5f703751efce5debe5f0674deb05c534375f670282f3f9533f2", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.12, 0.67, 0.21], "teacher_probs": [0.12, 0.67, 0.21], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8126220703125, 0.69140625, 0.109375], "student_probs": [0.12478030472993851, 0.5614838004112244, 0.3137359619140625], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.67, 0.21], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9b895812d92d22eeef1034d73a7e3acb0cfa0b472913e798206dc4d229a773f5:action", "state_id": "2d8d0b295ca00b67e10e64f2d329ff32dc8eb7b00a5582c871153efa3be2edf3", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.18, 0.63, 0.19], "teacher_probs": [0.18, 0.63, 0.19], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.771484375, 1.01953125, 0.08203125], "student_probs": [0.10702713578939438, 0.6416853070259094, 0.251287579536438], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.18, 0.63, 0.19], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "75687db71ef534feb405235cdd1c21c252e02aca52f21429149f0b10bf743614:action", "state_id": "f36d9dd87d152bc394a04dc65f68e1998b77abc80be8da2b01111d837a1752d5", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.59, 0.34], "teacher_probs": [0.07, 0.59, 0.34], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.388671875, 1.2578125, 0.76953125], "student_probs": [0.106690414249897, 0.5535852313041687, 0.3397243320941925], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.59, 0.34], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b2dc65b0aeecb4cdc58f576aa820fa97be5fa1dcb16a07b354bac30a1ad2792d:action", "state_id": "524ae377d3d7c7f3a8bf0741a0c702755bd4b3339601177e04157902c34fdd68", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.24, 0.74], "teacher_probs": [0.02, 0.24, 0.74], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.701171875, -0.08203125, 1.44140625], "student_probs": [0.08788342028856277, 0.1632286161184311, 0.7488878965377808], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.24, 0.74], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "07ebdc1d3515e280c2ab054c572a5e0831307528cc40952459d768568aa0ce35:action", "state_id": "3479c7a912724973a9ffd1d2bbf9ea62cb61245c212c8cdbe9892f7614e41704", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.06, 0.9], "teacher_probs": [0.04, 0.06, 0.9], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.01953125, -1.8671875, 2.599609375], "student_probs": [0.06968768686056137, 0.010562445968389511, 0.9197498559951782], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.06, 0.9], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "309223d47c132e5ad5784cf89f89870b5781ff6e5cf1d5da93c58fedc794d22d:action", "state_id": "cee75c38bc45b495c5f42b9d78a76a9e5ce301ef18e0f134ccc7f6dfda55c270", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.06, 0.89], "teacher_probs": [0.05, 0.06, 0.89], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.125, -2.4375, 2.51953125], "student_probs": [0.06589413434267044, 0.0065244026482105255, 0.9275814890861511], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.06, 0.89], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b2708cec5e5174ac0d471d3d5e2d63c363f0e886b5f7e983ed1dfcbb2321b0ad:action", "state_id": "7d3a96704d8afa4f6d3b2d3143bea3dc3627bdaa2645ca4bddb17fcaf9ee36e7", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.02, 0.95], "teacher_probs": [0.03, 0.02, 0.95], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.03125, -1.9921875, 3.740234375], "student_probs": [0.022428954020142555, 0.0031563465017825365, 0.9744147658348083], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.03, 0.02, 0.95], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c80e9d2e56489fc1c89764b41470942be9d507ef3f0a0ed4a833559aecbd6b3d:action", "state_id": "eeef3cd4e5defe6f85c079b0f76a233c776784adb643ea247859fb7b1725a973", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.56, 0.43, 0.01], "teacher_probs": [0.56, 0.43, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.33984375, 1.845703125, -1.63671875], "student_probs": [0.36908844113349915, 0.6121000051498413, 0.018811602145433426], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.56, 0.43, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1ff66ec14c3b86c895325405fb3abcc4dd949182d9421f9684bdd840a9f0157f:action", "state_id": "acb036c52478b523703d11b87b7268a246d094e3075ac22e8503a220dfa4c1cd", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.4, 0.58, 0.02], "teacher_probs": [0.4, 0.58, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.3046875, 1.283203125, -1.28515625], "student_probs": [0.2587682604789734, 0.6884540319442749, 0.0527777224779129], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.4, 0.58, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "083e614724b39215bc9b0929519fb89c907c58127fa7d9b6ef3890910a68047d:action", "state_id": "a9e745924d61b46d1b4c329295e7ae9f10ff0b0c01c7150f94ff1c600f21e71d", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.89, 0.07], "teacher_probs": [0.04, 0.89, 0.07], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.48046875, 1.05859375, 0.39453125], "student_probs": [0.04953287914395332, 0.6274721026420593, 0.32299497723579407], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.89, 0.07], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f40d90e81df246827e91bd4a484c5b1b3fa05886040e86010531ef287dea6646:action", "state_id": "1587809a7aabbeee0a59d7ab11e78f20ab402cd2d4cebc5173c7ce230995259e", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.79, 0.17], "teacher_probs": [0.04, 0.79, 0.17], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.328125, 1.328125, 0.3671875], "student_probs": [0.04832989349961281, 0.6883519291877747, 0.2633180618286133], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.79, 0.17], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9c9ff8add0fb535274ce3ec88c1be2766f03a4678a0a31f13174ce31e03a04f6:action", "state_id": "9d422ff3e8172e02ae535e2d54d406afa1c101bb17a5ffc9a02a998921f27ee3", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.72, 0.26], "teacher_probs": [0.02, 0.72, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.3828125, 1.294921875, 1.314453125], "student_probs": [0.03290427848696709, 0.4788258671760559, 0.4882698655128479], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.72, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f5f6289ec889b0e7ac55ba26cc89a69e4c353af92c8e0397375b9b2734f5a06d:action", "state_id": "7f3979a54d71c7ffbed11d9a74d971548bec50f3028d9e6fbd39478f55f833d4", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.59, 0.39], "teacher_probs": [0.02, 0.59, 0.39], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.421875, 1.482421875, 2.10546875], "student_probs": [0.018766706809401512, 0.3425375521183014, 0.6386957168579102], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.59, 0.39], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a75b33b0c8764bc738222d0ac476aca2b67891b2278dca8d3aeb3c03bd546cfd:action", "state_id": "588f6eb0c0f55d7b6281e32ce69c4e5798fa0b9c179ff73c7c6e9cd25c0a6491", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.02, 0.97], "teacher_probs": [0.01, 0.02, 0.97], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.83203125, -1.375, 3.62890625], "student_probs": [0.011344345286488533, 0.006591300014406443, 0.9820643663406372], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.02, 0.97], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a00b453e0046f10c0bf4a35af9e5fa665e42e863b0575bd7ba780296731e5128:action", "state_id": "9305a71a4197f7cd0ca3336376b34a6bee24273a3090be6bed8aa2b5899cd05e", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.44, 0.54], "teacher_probs": [0.02, 0.44, 0.54], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.1953125, -1.21484375, -1.421875], "student_probs": [0.07073532044887543, 0.5125579833984375, 0.4167066812515259], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.44, 0.54], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "73da471027e6e7fd39205cb21b246f2aa5097a8677776be6690ab62d226913fe:action", "state_id": "843fbffb73ba2a42b3c01634a87a6b0e78e22ddd7507ddd19f338ba7b48fac48", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.49, 0.49], "teacher_probs": [0.02, 0.49, 0.49], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.296875, -1.59375, -1.5703125], "student_probs": [0.08256017416715622, 0.45334452390670776, 0.4640952944755554], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.49, 0.49], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "901985b04c0ef5526651a13d0e9df0bd4f3ceb83783d0b07477f1ec1d22d7e7c:action", "state_id": "245d70414ad728a302348236625af9db89e7deb0924e99519552806e061036f3", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.22, 0.71, 0.07], "teacher_probs": [0.22, 0.71, 0.07], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.24609375, 0.46484375, -1.953125], "student_probs": [0.3108193874359131, 0.6327968835830688, 0.05638373643159866], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.22, 0.71, 0.07], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bb16db95f095df305671ff8f9d9b65ec335176d2ae4d1cfe52c3a64f516f6d65:action", "state_id": "7bb18c854dcd0cd2771ed5da51ff6fa263b91fa5f5b1857e2d841e457371e940", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.25, 0.7, 0.05], "teacher_probs": [0.25, 0.7, 0.05], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.5390625, 0.8125, -1.8984375], "student_probs": [0.4163450002670288, 0.547275185585022, 0.036379821598529816], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.25, 0.7, 0.05], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "315199effe1e1161b05bf4aeb44d14344473ba940f0ba7fa07b9afdfcc5f9597:action", "state_id": "554869e5488146678c8fddfc0e9d39c1d755bd43cca94ff705aa68ca378f2e69", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.6, 0.39, 0.01], "teacher_probs": [0.6, 0.39, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.98046875, 1.31640625, -1.1796875], "student_probs": [0.642188549041748, 0.3305703401565552, 0.027241067960858345], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.6, 0.39, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "98dcd68278344c3aa9a07625d918248072ad76cd02d56897cffeaad74989d03f:action", "state_id": "1569f386675e1ee5d7caee32d79eea64096cfbee2eb0101f316a056e25781d05", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.94, 0.04, 0.02], "teacher_probs": [0.94, 0.04, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [3.3779296875, -0.5615234375, -0.44921875], "student_probs": [0.9604021906852722, 0.018688324838876724, 0.020909501239657402], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.94, 0.04, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "025a5673d29fdb2d41f1e398917bb101ddd2773ce22c1c8d105f33fe7340caf0:action", "state_id": "ada35fd14e70e15fc16a6963b21076dd996ebdd67cb2ed519bc651e2a2009e01", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.6, 0.39, 0.01], "teacher_probs": [0.6, 0.39, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.21484375, 1.05859375, -0.501953125], "student_probs": [0.4914039373397827, 0.420320063829422, 0.08827611804008484], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.6, 0.39, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "add54a1b33632abdfc8313fc242ffc02d37bc69b86210e00e3e8de5313037088:action", "state_id": "26415e6d76a97715e4168b14ec42d3a80842ac10c37f0c7888e159472b002545", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.73, 0.24, 0.03], "teacher_probs": [0.73, 0.24, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.66015625, 2.154296875, -0.4677734375], "student_probs": [0.6072399616241455, 0.3661579191684723, 0.026602212339639664], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.73, 0.24, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a96a19a086b71d1c3fb9ff38672a2aa20572615f7a84c5bfd1021d646b5bed6a:action", "state_id": "a56e1a60b3e92cca29fcb41ea03a3bbab1b3a9c63ee1a957c7975e655a2d7a91", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.3, 0.67, 0.03], "teacher_probs": [0.3, 0.67, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.1875, 3.8642578125, 0.24609375], "student_probs": [0.06278267502784729, 0.9127271771430969, 0.024490198120474815], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.3, 0.67, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7dd55c16d7d9aeb4334621cf50c0d42f768f4d6b1f3b18a22f13260354aa2e7f:action", "state_id": "47f24a0815b5394468dff939f6458ea7ec2fe5e83f3e621004836144844f3d8e", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.32, 0.5, 0.18], "teacher_probs": [0.32, 0.5, 0.18], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.20703125, -0.0390625, -0.322265625], "student_probs": [0.061254534870386124, 0.5353959798812866, 0.40334951877593994], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.32, 0.5, 0.18], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9c938e5f11da55290ad3b02459161b79029a13a9e5fae2a81147598acaf7c85c:action", "state_id": "cf3afcd11a11c20a6df11936a6f3ec03d11e37ce8d1c70313aad36c1bd0772ae", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.31, 0.42, 0.27], "teacher_probs": [0.31, 0.42, 0.27], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.4765625, -0.1953125, -0.4794921875], "student_probs": [0.05507715418934822, 0.5391452312469482, 0.40577763319015503], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.31, 0.42, 0.27], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "628a971b0c55ce25a4dc5588b44b037d988db43434dc25ff6c3478aeba00dd1d:action", "state_id": "90ad02c2359a89bc7f483587d44616165c982bae24e0f2016754bedc247c183f", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.26, 0.16, 0.58], "teacher_probs": [0.26, 0.16, 0.58], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.8359375, -1.7734375, -0.986328125], "student_probs": [0.22710612416267395, 0.2417532205581665, 0.5311406254768372], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.26, 0.16, 0.58], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d9a6f9926b8b21626d3a57a606abab1c1b5c7103425ee52eee4d5001d9dfa2c6:action", "state_id": "03e37794e440a57bce967f1c46ef2b6aaa0a332c9296cb533500bb97253937be", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.11, 0.73, 0.16], "teacher_probs": [0.11, 0.73, 0.16], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.984375, 0.3671875, -0.59814453125], "student_probs": [0.024740390479564667, 0.7062714099884033, 0.2689882218837738], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.11, 0.73, 0.16], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "893f1ca3fca8aae624efb364e7cd7d762976ed778a75044900e606719ae63c55:action", "state_id": "29954521fcf9944f7d3642a09aafe2b69c14fddc146c7da1af6bf265d13182b2", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.68, 0.23], "teacher_probs": [0.09, 0.68, 0.23], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.98046875, 0.91796875, -0.255859375], "student_probs": [0.015249534510076046, 0.7521881461143494, 0.2325623482465744], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.09, 0.68, 0.23], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3ddd986cadfd21f52df35f3a0bc3b35a5e504e72ff7948fd16d102685a6ea718:action", "state_id": "42e5352bf74f6b2248b5930b0e56145856c9f5ed854e76d381f8a1dfa9d89591", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.67, 0.28], "teacher_probs": null, "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": null, "student_logits": [-2.51953125, 0.74609375, 0.43359375], "student_probs": [0.021569279953837395, 0.5650392174720764, 0.4133915305137634], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": "Rounded target has no identity probability-simplex representative", "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": null, "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b88fbb721a40c83f6c2e02e331dc2d3e39edbb47c9eab1514916ba0e5c8242ad:action", "state_id": "2854f2b205f3076f9742b51b292b878ab4315c511b6d36b426ac9b556a776eb1", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.23, 0.75], "teacher_probs": [0.02, 0.23, 0.75], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.10546875, 0.48046875, 0.87109375], "student_probs": [0.011059432290494442, 0.39910364151000977, 0.5898369550704956], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.23, 0.75], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4a73217a5288e3e4e9493a98e76ce97f2d9981d7e37dd17173860d956f343de9:action", "state_id": "6e364a63d471a90d4a68b7776ea65c9cc9866200971261d9529a21c7817d375a", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.01, 0.98], "teacher_probs": [0.01, 0.01, 0.98], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.6171875, -0.14453125, 4.552001953125], "student_probs": [0.019004354253411293, 0.008872436359524727, 0.9721232652664185], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.01, 0.98], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bce227fbb64be7c6ccb8f126353434220793127c19b2a8f9bfbc021a347e089a:action", "state_id": "353810ac56d30db59ab0ebac6edb200e27f6ae82d36bac1299d54153a4b4df02", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.35, 0.65], "teacher_probs": [0.35, 0.65], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.16796875, -1.3828125], "student_probs": [0.3132096230983734, 0.6867902874946594], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.35, 0.65], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a3d03434d94211df660a3f206237a01be5ea71838fc8490586c730a78987a5c1:action", "state_id": "2334afd2c970cad3eeaf4dbb85d9ef10d7aa5c26c7fe37c34b9a870938837c2c", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.09, 0.37, 0.54], "teacher_probs": [0.09, 0.37, 0.54], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.77734375, -1.4453125, -1.56640625], "student_probs": [0.27558597922325134, 0.3841107189655304, 0.3403032422065735], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.09, 0.37, 0.54], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "696751b0530da35591624644068a3b016164d2b1a5d7e0a756b002cf6a6e9020:action", "state_id": "5f4013b185a0e13f6eaf645dcbc30a6345aa639f621cf47d3464846594c7ad39", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.36, 0.64], "teacher_probs": [0.36, 0.64], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.6875, -1.6796875], "student_probs": [0.4980468451976776, 0.5019530653953552], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.36, 0.64], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3f38df5ac1889f832eb0da777f7ad9d5976d0d2fbe73f226c9d912732ae7dc2e:action", "state_id": "9e076243c3f1fcfe6f562788d76ab5041657fda059b458e53c1e737114374d66", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.19, 0.22, 0.59], "teacher_probs": [0.19, 0.22, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.640625, -1.421875, -0.98828125], "student_probs": [0.24012164771556854, 0.29883623123168945, 0.4610421359539032], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.19, 0.22, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ad73b5af47ba472452cdf800219ad3899aad3d86ad9052c05c53ca0eb0852afd:action", "state_id": "d322f0ca8af668252287369d196a059a3d9e46f5c085ead1cbc521c7fdf96112", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.58, 0.42], "teacher_probs": [0.58, 0.42], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.93359375, -2.6484375], "student_probs": [0.671470582485199, 0.3285294473171234], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.58, 0.42], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "89ceeb6b712265f026afcb77692d8510acace01e5cc840fb0b11354419161c6d:action", "state_id": "d5b701d8e32fc1777f1aadc92099668d41441bfd41a2b587fbdc14d38235779a", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "west"], "gold_index": null, "teacher_raw_probs": [0.2, 0.8], "teacher_probs": [0.2, 0.8], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.55859375, -1.46875], "student_probs": [0.2516476809978485, 0.7483522891998291], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.2, 0.8], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c3e83c68ba40358c763591de2a1dd8ec78e487f4bca911b3f16df5f48e61c195:action", "state_id": "0a20b94d7f7bfc02a569e4e2b3a0065700d99b776ec37b0cbbc0a8bc4be52ec2", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.42, 0.58], "teacher_probs": [0.42, 0.58], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.0546875, -1.6796875], "student_probs": [0.4073334336280823, 0.5926666259765625], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.42, 0.58], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7d7c9b68741e1bf0606929bf1512cda6cceb96c471d71e68998bd10309256aaf:action", "state_id": "b1cf957d5ac7cf8fd6049395454ff52628b30564adb7ffdf843b5fc5488b6eaf", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.17, 0.5, 0.33], "teacher_probs": [0.17, 0.5, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.77734375, -1.609375, -1.7265625], "student_probs": [0.14133599400520325, 0.45445945858955383, 0.4042046070098877], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.17, 0.5, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "988c066625d7794b2238bc5a9963cd13f1a19d9f6b730b975ce74fd6e49d2a5b:action", "state_id": "f1dbf89808caca6cd31e0eb67fa08d22fae08ea6e48c198df401c84d59a6a9bb", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.2, 0.8], "teacher_probs": [0.2, 0.8], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.17578125, -2.26171875], "student_probs": [0.5214711427688599, 0.47852882742881775], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.2, 0.8], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b12b4229fb199d4ddf76d50f582b2cd113e1ab793395feeb581302ddc916144a:action", "state_id": "a8bfb6cc1ca65d874e9d7e76215835d4422f0d091e2ab80b5437ba80e8ea676e", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.16, 0.79], "teacher_probs": [0.05, 0.16, 0.79], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.5078125, -2.12890625, -1.05859375], "student_probs": [0.1487990915775299, 0.21734876930713654, 0.6338521838188171], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.05, 0.16, 0.79], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9c3c9aba7a04ca9626c30589415427eba8d977947aba062ecca2308f5e10f3b2:action", "state_id": "2f1745ae7e7f6ec2be79f2e84654b389bdd49033f4266e9c74af6bf488a6e118", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.22, 0.31, 0.47], "teacher_probs": [0.22, 0.31, 0.47], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.85546875, -1.9296875, -1.48046875], "student_probs": [0.29555544257164, 0.27441391348838806, 0.43003061413764954], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.22, 0.31, 0.47], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d312424ff5d4722490d775b964e1bfd1847c96b3239c7c1fe38e2e3f9529dc35:action", "state_id": "64146620f842064d30fe91f090fc83425a5cc5776904861796b52138c1dbbfc5", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.16, 0.58, 0.26], "teacher_probs": [0.16, 0.58, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.7109375, -1.2578125, -2.01171875], "student_probs": [0.13719984889030457, 0.5867293477058411, 0.27607080340385437], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.16, 0.58, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e562ed2690815f118b85a05a653cdb0f876b7b7a24c88b64cf9e48ff00da7d0b:action", "state_id": "ad0b4be62c526337a04bbe6c38c688f1076c27daa99fc6121bd9e4cf8fa39ad2", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.31, 0.69], "teacher_probs": [0.31, 0.69], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.96484375, -2.44921875], "student_probs": [0.37387582659721375, 0.6261242032051086], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.31, 0.69], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1519c00d9e13910afa165f4b38054ae5dc4cd63574d21880a114a4f1dc114d8a:action", "state_id": "2d10663b05dd377623c65a6bc1f253ad36194c4a2861869ace778c95f09aaafc", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.3, 0.23, 0.47], "teacher_probs": [0.3, 0.23, 0.47], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.1484375, -2.3515625, -1.328125], "student_probs": [0.24465514719486237, 0.19968172907829285, 0.5556631088256836], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.3, 0.23, 0.47], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "59f6432af190635d0371aa9c303ec0b47cee63d06d90212fc044d1da9c32a193:action", "state_id": "eae06654541b0200113b87f9621b7b3f7da6fb8a236423d4337ffc68df231f9c", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.27, 0.28, 0.45], "teacher_probs": [0.27, 0.28, 0.45], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.2578125, -0.4716796875, -0.66259765625], "student_probs": [0.4041097164154053, 0.32630065083503723, 0.26958972215652466], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.27, 0.28, 0.45], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e5d54384edc08e2d08c292e0206854743b1ad73e2f440b8fdbb9c918edc17db2:action", "state_id": "5028c03450fcb5addf1364fc93b7e51fef9f31bd43526f94438bf3a0aabb0a49", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.46, 0.41, 0.13], "teacher_probs": [0.46, 0.41, 0.13], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.76470947265625, -0.8984375, -1.103515625], "student_probs": [0.3864811062812805, 0.3381044864654541, 0.2754144072532654], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.46, 0.41, 0.13], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2722b9ff5e4236209820df8127f1c0ad014555585309bb7158b5e13968a6d763:action", "state_id": "0d3a9dcfce0288e9a364e1723a669524d8a7f32b937924e5054d65bf9745a219", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.37, 0.5, 0.13], "teacher_probs": [0.37, 0.5, 0.13], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.85205078125, -1.42578125, -1.099609375], "student_probs": [0.42659854888916016, 0.2403540313243866, 0.33304741978645325], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.37, 0.5, 0.13], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c7554c880da2053b39dbd443573193ab0c864924927a2c7ebfa8a7fb6cf81e14:action", "state_id": "3139e4498abac81a5e749d695ed1a424e9ace61196577b3e580ba41b35126379", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "west"], "gold_index": null, "teacher_raw_probs": [0.51, 0.49], "teacher_probs": [0.51, 0.49], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.9033203125, -1.056640625], "student_probs": [0.5382551550865173, 0.4617448151111603], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.51, 0.49], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b68fa8a2e867c559e74182ac6985dd6fd538ab72578439c0d463a0b4079693f6:action", "state_id": "7b636e3ff2cddeccaf32aa26c5e97cf63f8f000d5171b57145f9e7a075615ead", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.34, 0.66], "teacher_probs": [0.34, 0.66], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4853515625, -1.33984375], "student_probs": [0.701508641242981, 0.29849135875701904], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.34, 0.66], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c165cced3c79d593d53d9db9be95ce069cb6d30049a61d799145fd094b02a317:action", "state_id": "3e33fa096c2a7178e0460e2beaf72c248462df26db8d029ce03725f606561ee5", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.72, 0.11, 0.17], "teacher_probs": [0.72, 0.11, 0.17], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.10546875, -1.25, -1.80078125], "student_probs": [0.6658166646957397, 0.21197812259197235, 0.1222052350640297], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.72, 0.11, 0.17], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5f286dbe4dcf8f6db7e09af462c36abdb55a161e135c1a04840b240fe2a39209:action", "state_id": "1bfd3860988736091ee5d0b0971fc8a93720da8ac9a35ae9e90d444ffadee501", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.43, 0.57], "teacher_probs": [0.43, 0.57], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.21484375, -0.341796875], "student_probs": [0.29462069272994995, 0.70537930727005], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.43, 0.57], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c2197f4f56adb64d2f37d6f8eb90cc09de0a451cf042a5caaecbd2542d30fb4c:action", "state_id": "022f4ca937f5b254a6c9429478eacf2ccccf1a4a2e75ab64c7edad43b11d02fa", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.29, 0.71], "teacher_probs": [0.29, 0.71], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.49609375, -1.109375], "student_probs": [0.19993209838867188, 0.8000679016113281], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.29, 0.71], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "56dd417aa437b62c650555f58cac669aa76564d27654a50e5778d1f86e2351c7:action", "state_id": "387286ba9574c9515d48a29509332d54c3b33163022a27489fc0329607ec4828", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.24, 0.76], "teacher_probs": [0.24, 0.76], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.6015625, -1.041015625], "student_probs": [0.17356820404529572, 0.8264318704605103], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.24, 0.76], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8cb66f3cf2a16d667eaeeaaa4b67b555856570386a95dbabc93c4832293c92a6:action", "state_id": "660025ce7c0c28cd08cf124cd0767ec04029bd01ba201bfb3daa638817c63967", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "west"], "gold_index": null, "teacher_raw_probs": [0.47, 0.53], "teacher_probs": [0.47, 0.53], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8408203125, -1.142578125], "student_probs": [0.5748721361160278, 0.4251278042793274], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.47, 0.53], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d75f2e5296e872c3ed0efdb485467a12b054f5db2b605196dd09f8c20ea029a8:action", "state_id": "2b98e7b431c4dc0bc9d8c8042f5dcc1579e21478792330deb5736694050d1e7a", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.5, 0.5], "teacher_probs": [0.5, 0.5], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.177734375, -0.8935546875], "student_probs": [0.42942941188812256, 0.5705706477165222], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.5, 0.5], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "45e4d393245128be639fb1cc711ca52c409accb58b07f9e09587b3f7dbc3c148:action", "state_id": "d9f4fb0613284922a9c382f19decee1b077e52fd56a3232b0a75cadcfdfcd097", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.36, 0.64], "teacher_probs": [0.36, 0.64], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.421875, -0.9033203125], "student_probs": [0.3731902837753296, 0.6268097758293152], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.36, 0.64], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7fdf580f671884dde95bfb942c7a46f98296574e53d0d715c3c37a3379d314f7:action", "state_id": "13f7ae9bb340050aa4ec4fd17f4f74b9215ff22e66dfda821a4af644651a8aac", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.27, 0.45, 0.28], "teacher_probs": [0.27, 0.45, 0.28], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.27734375, -0.98828125, -1.044921875], "student_probs": [0.278022825717926, 0.37120917439460754, 0.3507680296897888], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.27, 0.45, 0.28], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ad84014b7966326d54b24bce9baa9b8b73358f84e89910d377ac052c8673aa43:action", "state_id": "cbb6544c8fc94b881658dc41a5fd508de16458c2352199f9ba01c6c90f9601a5", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.27, 0.73], "teacher_probs": [0.27, 0.73], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.77734375, -1.48046875], "student_probs": [0.4263215959072113, 0.5736784338951111], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.27, 0.73], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "62561b8c03b4bdfe9d177f93090329cc8f7709a6e47b890afd08490292c00d01:action", "state_id": "2d9c7ba5ffc346fd6e59a12527864a688211c5e32de8afa88e1ed3a14993f95a", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.15, 0.64, 0.21], "teacher_probs": [0.15, 0.64, 0.21], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.3828125, -0.884765625, -2.28515625], "student_probs": [0.32774826884269714, 0.5393111705780029, 0.13294056057929993], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.15, 0.64, 0.21], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8ad8cd123bd89249521142276fd0838ae4ebc59a324c02b8f94107a25dee06cc:action", "state_id": "5c961acecbde35294dff94f5c66688ae319c70f3b2d70297a3c8159c0a676bbf", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.41, 0.48, 0.11], "teacher_probs": [0.41, 0.48, 0.11], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.037109375, -1.16015625, -2.25390625], "student_probs": [0.4586315155029297, 0.40553218126296997, 0.13583625853061676], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.41, 0.48, 0.11], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "53c3d04e4fdc09225585e3b30459db682f5d002a631422074266358a49799534:action", "state_id": "067a33fc5f5cc311b48f00eecc6070b72f5f556a57d171317f87060029f2a5d9", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.62, 0.22, 0.16], "teacher_probs": [0.62, 0.22, 0.16], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.59375, -2.0859375, -2.66796875], "student_probs": [0.5120714902877808, 0.31302300095558167, 0.1749054491519928], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.62, 0.22, 0.16], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0abf8e2f6b96c489c339acdc644dff59947b9e08d55a6ca74bc1ce20d2a6eae1:action", "state_id": "05aa967c4b4a0091cce22cbde3fa136a8ed73f880a5888bf4ba1dbc7a90714a3", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.51, 0.2, 0.29], "teacher_probs": [0.51, 0.2, 0.29], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.33984375, -1.859375, -0.5537109375], "student_probs": [0.26387375593185425, 0.1569519191980362, 0.5791743397712708], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.51, 0.2, 0.29], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ac79aacdc0e4f91935303dbb00b502888aaf330139fee5b585671bf5ffab59c1:action", "state_id": "95e677c6a5b0582f1a64c68f85704f41fa8fba9a4822e23f5f014946d4c74525", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.41, 0.59], "teacher_probs": [0.41, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.7109375, -2.6171875], "student_probs": [0.7122321724891663, 0.28776779770851135], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.41, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ef3e0632ca3e4b7be7333f9b5d3d7fe3ea204df6d8df67156c5842dc43cba587:action", "state_id": "bb4855d75e535079ae052852602a958179edfb08a7ca166abc09c266884e6326", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.67, 0.18, 0.15], "teacher_probs": [0.67, 0.18, 0.15], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.70703125, -2.1015625, -1.85546875], "student_probs": [0.3943140208721161, 0.2657660245895386, 0.33991992473602295], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.67, 0.18, 0.15], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e861d250e563fb62f012a0460b54e533095774b10fcf59d7be67ab113166270a:action", "state_id": "66de2e5169fde40b62bf3d0df71f44081d70b3128058639ba9f1245ac3f6e907", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.2, 0.55, 0.25], "teacher_probs": [0.2, 0.55, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.51953125, -0.515625, -1.80078125], "student_probs": [0.2230270802974701, 0.6086232662200928, 0.16834968328475952], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.2, 0.55, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "12f261d8045f5b1713ecee6f8fc1df24cebb267aa87552397e8bd0d20174b727:action", "state_id": "bec1f19e656aa4858661dcd5ffebca57c2afeae87bcd113eb0c531ac56dcf4ee", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.15, 0.24, 0.61], "teacher_probs": [0.15, 0.24, 0.61], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.84765625, -2.609375, -1.48046875], "student_probs": [0.34357979893684387, 0.16040480136871338, 0.49601536989212036], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.15, 0.24, 0.61], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "794a8f5447673c4ede33938d53a23ecfd0878a0ed0eff12a9dd9bd4459b06841:action", "state_id": "02f0b4785e64e8b5b7238f12c0be82de0176534ca86b385bfc1630d5e004057d", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.61, 0.39], "teacher_probs": [0.61, 0.39], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.53515625, -2.59375], "student_probs": [0.7424216866493225, 0.2575782537460327], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.61, 0.39], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6ba167a17eca7444dc335a62a95465488e9568fab6edd8f0be2a05984fed7c1e:action", "state_id": "89d509b95bcda1270f74e23aa7ab509f0b4a7128eca008079b032d97403df1d5", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.1, 0.76, 0.14], "teacher_probs": [0.1, 0.76, 0.14], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.078125, -1.8671875, -2.64453125], "student_probs": [0.16950812935829163, 0.568976104259491, 0.2615157961845398], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.1, 0.76, 0.14], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "dd2cd8a521e097a1dbf6a54fa5e223ef73abdc1c3c380ca8035445de3614cf47:action", "state_id": "32f2ef6db233f7f6f6a2d5970140840709ecf8b281064353c356c99688b4aa47", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.4, 0.6], "teacher_probs": [0.4, 0.6], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.013671875, -1.55078125], "student_probs": [0.6311396956443787, 0.36886027455329895], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.4, 0.6], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ec8de3e42ca375a3a030537df944c1fb254854a46d81414f27dc9297902588e5:action", "state_id": "3daf543a9035a036ae839809873ea952585035f560a7338e2888e62a80a454a5", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.28, 0.18, 0.54], "teacher_probs": [0.28, 0.18, 0.54], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.18359375, -1.2265625, -1.51171875], "student_probs": [0.373383104801178, 0.3576790988445282, 0.26893776655197144], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.28, 0.18, 0.54], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c28ef4d9cd135b26017fe610f0ccdaa7025d58210e48234bcafa5da25fd3b835:action", "state_id": "d6aff1905fed63566e28a545e2d5e9c0c614ac77014adadfac1efb14a81b43eb", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.38, 0.27, 0.35], "teacher_probs": [0.38, 0.27, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.0087890625, -1.34375, -0.6376953125], "student_probs": [0.3159871995449066, 0.22604651749134064, 0.45796623826026917], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.38, 0.27, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "662e56b5e9f2ab172cab022852219d5c74735bd66d25e2bf62c3b1df10ffda44:action", "state_id": "0425785a7bdab595c7f8f27a49df7d48614cbfc91bd484e6a9626472364852c1", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.72, 0.28], "teacher_probs": [0.72, 0.28], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.150390625, -0.7539520263671875], "student_probs": [0.40216830372810364, 0.597831666469574], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.72, 0.28], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f9e005f0a8552671300d1b57d1890972a39f4ed3086b519a8ec6a4c501ee54bb:action", "state_id": "6a45ececf841f15316a5f1d747a351f79257bb867212c45cde2cc47c36f243f6", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.49, 0.18, 0.33], "teacher_probs": [0.49, 0.18, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.50390625, -1.515625, -1.46875], "student_probs": [0.330673485994339, 0.3268209993839264, 0.34250548481941223], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.49, 0.18, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3f511cd1ff9924cc1366b8d887965151eb42d18a822ca80563284ba189623c60:action", "state_id": "b8fddd8c8abe50bee76ce31be5f2b9fc4152d85f68c1c8d882f4284204f6700f", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.14, 0.43, 0.43], "teacher_probs": [0.14, 0.43, 0.43], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.55859375, -0.84033203125, -1.359375], "student_probs": [0.23412001132965088, 0.4801485538482666, 0.2857314944267273], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.14, 0.43, 0.43], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bdc635b016ed18a3c8f26b6947080d493cc7cf58ea1dd8fdb5425ce89cea54ce:action", "state_id": "a701c9984e3d20f91b5102c416b24f4b0d8103a35e104f26ca268cc7e7009a7e", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.25, 0.26, 0.49], "teacher_probs": [0.25, 0.26, 0.49], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.784423828125, -1.26171875, -1.28515625], "student_probs": [0.44912606477737427, 0.2786645293235779, 0.2722092866897583], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.25, 0.26, 0.49], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "af30fbe42c7373348106b49d84ff72ef4c5af2a5deae35c32ceafc70cc7f0f52:action", "state_id": "87ff1f06a562ae737fb3fb16d4fb1222aa4a8ff2755e2396cf72eb590f445746", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.18, 0.56, 0.26], "teacher_probs": [0.18, 0.56, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.6875, -1.33984375, -1.390625], "student_probs": [0.2658589780330658, 0.37638866901397705, 0.35775235295295715], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.18, 0.56, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3547a96d919e3198adee21661f56d2aea2a448b51d43ff5873256d27736032d2:action", "state_id": "dacdc3361e1f384ae0a9d8bc206e33db874bfb57fac93687c986741dfeb8f355", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.18, 0.51, 0.31], "teacher_probs": [0.18, 0.51, 0.31], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.37109375, -0.62109375, -1.064453125], "student_probs": [0.2234211266040802, 0.4729825258255005, 0.3035963475704193], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.18, 0.51, 0.31], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2bb50e71bf69aafe136278e865720f45c2b680bfdfacc39e56db9a051b475c45:action", "state_id": "235087febbd55d863f5ab84598b624c8d8ce93763de308e3f5cf91baede05d2a", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.26, 0.55, 0.19], "teacher_probs": [0.26, 0.55, 0.19], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.659912109375, -0.8232421875, -2.09375], "student_probs": [0.47899529337882996, 0.40681588649749756, 0.11418876051902771], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.26, 0.55, 0.19], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5e5dd3f5e1782cd2ec9a9f4b4ceb0acbfaed10e8dba37f7c926ee4b5971f5573:action", "state_id": "75c0daa77d7b530e148f353be814615753fbfe2ecebc4ac25e61583a67bb98e8", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "west"], "gold_index": null, "teacher_raw_probs": [0.3, 0.7], "teacher_probs": [0.3, 0.7], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.28125, -1.185546875], "student_probs": [0.47609245777130127, 0.5239075422286987], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.3, 0.7], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fb6462a7bce85758645ce65f4a9293ba21aa3776b7d7c0f3685ed96f8f5250d5:action", "state_id": "c99f6cd89c076497de8e291351a9b1e9cb4bb2403fd7e9815d9b7188fcbf3d7b", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "west"], "gold_index": null, "teacher_raw_probs": [0.65, 0.35], "teacher_probs": [0.65, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.48046875, -2.453125], "student_probs": [0.7256485819816589, 0.2743513584136963], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.65, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "26ec6bc46128e0667946df15f7779664a87f747cd024c87ef59b2aff085d5e95:action", "state_id": "dfd59c6aebecfcadc29bd4a387501be2421fb15ce070d3bf342226c9cd0ce97e", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.56, 0.09, 0.35], "teacher_probs": [0.56, 0.09, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.587646484375, -1.78515625, -0.84375], "student_probs": [0.48169392347335815, 0.14544515311717987, 0.3728609085083008], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.56, 0.09, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cd431293b479d32f7fda94d1169e2d50c04fa824034c4bc20d28ba0b22a1f5d0:action", "state_id": "8adde5130bf77ebeceee6858532ec16d895aa63057856a0e4e9f4b36ba0453f5", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.42, 0.58], "teacher_probs": [0.42, 0.58], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.3203125, -1.013671875], "student_probs": [0.4239349663257599, 0.5760650634765625], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.42, 0.58], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "58a6da2e834b8f5c041efeb19ca751fc66d4f1d96710cf7d30e6eebf7a492307:action", "state_id": "25eb7cedd3730815be83b00c0221065a4aac9615804c000eb3698611d64d718c", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.34, 0.66], "teacher_probs": [0.34, 0.66], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.1953125, -1.3125], "student_probs": [0.5292633771896362, 0.4707365930080414], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.34, 0.66], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3c80a6fcd99983c8eecf3227ed2202d7dac030e1a7d8b2f2bd8014078c083ed1:action", "state_id": "17ce6331775fab8864fc213808ceba1f836f78c53c48be2391e94eb92a0ad14d", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.29, 0.48, 0.23], "teacher_probs": [0.29, 0.48, 0.23], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.41796875, -1.19921875, -0.8388671875], "student_probs": [0.2482033371925354, 0.3088940382003784, 0.4429025948047638], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.29, 0.48, 0.23], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9e913840bfa976e35f9c6d22c4be0a4086a8bc862865e7b07ea9e32a7fb2c88e:action", "state_id": "2a1edffd89377b2166d408d019611f98e868750dab1c61ea74e788299cee87d8", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.48, 0.52], "teacher_probs": [0.48, 0.52], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.146484375, -1.1328125], "student_probs": [0.4965820908546448, 0.5034179091453552], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.48, 0.52], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "350c4ddfabb802723b3f5f9274f1f18f1921a98fccf37aa1d5bbb32a0de8cdbb:action", "state_id": "276c1cc927e09d2c9a0ac2591f2c6a5377d890012b53b87c484c54c9fb55165a", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.54, 0.21, 0.25], "teacher_probs": [0.54, 0.21, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.40234375, -1.5078125, -1.515625], "student_probs": [0.35806331038475037, 0.3222220838069916, 0.3197145462036133], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.54, 0.21, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "28fc62edfc4e349f0ccdbc7f1d5167e523d198979bee6c5dac6d531a7e5b4ca6:action", "state_id": "5ac205873a45d24a8c2c5f4b2ce443539b10304ac0fa429aab0604b09290cee9", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.52, 0.48], "teacher_probs": [0.52, 0.48], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.09375, -1.28125], "student_probs": [0.5467381477355957, 0.4532618224620819], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.52, 0.48], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7a99331f7a5fb8e659f35ec0a4ace3b4fb22ce9008019ddbfb7251f3da8e9b0f:action", "state_id": "0dd532f863b27b11eccf7c533e2ff5c258126b54d730a19f761e14f183bea6b0", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.43, 0.57], "teacher_probs": [0.43, 0.57], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.0546875, -1.107421875], "student_probs": [0.5131805539131165, 0.48681947588920593], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.43, 0.57], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "46c574df5665788e80cecf04ddeda994fb0cf2ae1037b13b0679687b0963aa2b:action", "state_id": "7add8ba8f9f390098d58571e0787bab161debf8ba60d52b408e19864a451bee3", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.43, 0.25, 0.32], "teacher_probs": [0.43, 0.25, 0.32], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.55517578125, -1.65625, -1.10546875], "student_probs": [0.5237537026405334, 0.17415527999401093, 0.3020910918712616], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.43, 0.25, 0.32], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c363dd2740411e3020e15a6c91557cd4531598cece8adaa68c68b46008aa8bc6:action", "state_id": "ea6b93f412391606ea802fb8c042f5a4aa9f52bff908f79219e106e2e9c51dcb", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.4, 0.27, 0.33], "teacher_probs": [0.4, 0.27, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.103515625, -1.6171875, -1.78125], "student_probs": [0.4748200476169586, 0.284082293510437, 0.24109753966331482], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.4, 0.27, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3bd95a80eb5b07721d0dffee4dd3c2af1ba86362eb6b1e5f318dde0d95a05452:action", "state_id": "fed125836aa57a38bec2fdf0f6add36edfd3af6f06aeb6da55d8c268ffc79783", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.28, 0.27, 0.45], "teacher_probs": [0.28, 0.27, 0.45], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.5576171875, -1.62890625, -1.8203125], "student_probs": [0.615211546421051, 0.21075095236301422, 0.17403751611709595], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.28, 0.27, 0.45], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "dc298ef17802a082948577100e60518316eec79b42e7c400532ca366ae57c053:action", "state_id": "5b0d9e77ea6a80f8481b2ab020124024fa9e8fb4a365a7771b674e84ed0a7519", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.5, 0.28, 0.22], "teacher_probs": [0.5, 0.28, 0.22], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.28125, -0.66162109375, -0.87646484375], "student_probs": [0.22950421273708344, 0.4264735281467438, 0.34402230381965637], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.5, 0.28, 0.22], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "77c8524466e78d742a4f8a78f698d4ac265ec1f597994f7203d431265c41ccc3:action", "state_id": "0e9aafd60227d3828856f86e2cfa751b4d4e5b0dd117b143ce1376d370139134", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.19, 0.25, 0.56], "teacher_probs": [0.19, 0.25, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.89208984375, -0.03515625, -0.4453125], "student_probs": [0.20328544080257416, 0.47892531752586365, 0.3177892565727234], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.19, 0.25, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7453dc42b626f4e2f540e347d2dcb6c40a85e0b5c347c12892ab5f1ad711fdc5:action", "state_id": "e3fcffe6ea7f3bced5e6c948c9803da487e6309059678d48d9ef82207c0fc418", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.17, 0.26, 0.57], "teacher_probs": [0.17, 0.26, 0.57], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.09375, -0.11328125, -0.819580078125], "student_probs": [0.20075711607933044, 0.5351593494415283, 0.264083594083786], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.17, 0.26, 0.57], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5ea330678b84d9ea2cab23f06353fc904c3b5af48f51dfaa150ebf09f52dd8df:action", "state_id": "1ce625d981aae2fc20cb68af1356b021c6fcb2a09cf8d84569d642326f65f208", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.64, 0.31], "teacher_probs": [0.05, 0.64, 0.31], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.01171875, -0.77783203125, -1.62890625], "student_probs": [0.1694640964269638, 0.5820333361625671, 0.24850265681743622], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.05, 0.64, 0.31], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fc42601501993dc1594e6c0bdc79f64d51b12fd3d87e09f43fc474bc5d8a906b:action", "state_id": "1bea1467cd02d06c139c47596120d61f6769f83451a1ec9cecfbcad92e77e7be", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.13, 0.7, 0.17], "teacher_probs": [0.13, 0.7, 0.17], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.6953125, -0.57666015625, -1.203125], "student_probs": [0.17554277181625366, 0.5372884273529053, 0.28716883063316345], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.13, 0.7, 0.17], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4c789a13b69c0a8fbd37ab0689b6e43880dd5eed74708a35d789a51f4823f168:action", "state_id": "634cb4762599b49710d99973a06ea82156db3982509af0746f280c5c0a6d4c47", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.36, 0.49, 0.15], "teacher_probs": [0.36, 0.49, 0.15], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.546875, -1.1640625, -1.42578125], "student_probs": [0.27815374732017517, 0.40788552165031433, 0.3139607012271881], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.36, 0.49, 0.15], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "14ca0ad30bf6f4ece3baa27886b8a04adf6760bde6a05051e70067128684a1d9:action", "state_id": "4a4d2616e2cb060bfa71491c224d772f4a9481b4359a682fd8510d153f76afd2", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.28, 0.23, 0.49], "teacher_probs": [0.28, 0.23, 0.49], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.82421875, -1.83984375, -1.7578125], "student_probs": [0.32752978801727295, 0.3224519193172455, 0.3500182330608368], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.28, 0.23, 0.49], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b9288887a1b2acf11a8f605e56732cb64e7f5b7abc74e2d52de5667e906b9eca:action", "state_id": "1f44c16e9397eb3759a63a511a224e6d7f95934f7bf97a94a6ab582d6cab3fca", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.51, 0.49], "teacher_probs": [0.51, 0.49], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.390625, -1.4375], "student_probs": [0.5117166042327881, 0.4882833957672119], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.51, 0.49], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2c9eb36da84bd59a6e1482b83af98116a11b96f6ada6fe4255fd47daac94e716:action", "state_id": "4234a9e6b995d098658eb64f54236f74e690602476869d340bc7760436879ce0", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.37, 0.63], "teacher_probs": [0.37, 0.63], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.076171875, -1.013671875], "student_probs": [0.4843800961971283, 0.5156199336051941], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.37, 0.63], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9b59273977dbaa006014925afc6ccb8abbe9a78626a196ff658bd5c41864640c:action", "state_id": "bbd96f10d907991cd8a0ab78d76c58e21a8cfd181b0805445e8b0959ffc94ef7", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "west"], "gold_index": null, "teacher_raw_probs": [0.24, 0.76], "teacher_probs": [0.24, 0.76], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.33984375, -1.12890625], "student_probs": [0.4474602937698364, 0.5525397062301636], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.24, 0.76], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e40d6d32c2ec938260f227c7078ecda43574fbe36b607104f3e9859c2be8a482:action", "state_id": "27d831d55e56f1e56b62699a686619d8949e9e1e4fbe054e1572b026f8da5908", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.46, 0.54], "teacher_probs": [0.46, 0.54], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.31640625, -1.65625], "student_probs": [0.3407045304775238, 0.6592954993247986], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.46, 0.54], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "39e3f7e6519ab82be802e0dc4b3a87f2f5caf74cffe140e2fef0315c6832ace1:action", "state_id": "dda58fbd6828f25b1cc3fb965172cd852f494ae9f4628f432475a0b19c634563", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.68, 0.23], "teacher_probs": [0.09, 0.68, 0.23], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.98828125, -0.4345703125, -1.060546875], "student_probs": [0.12109821289777756, 0.5726718306541443, 0.30622994899749756], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.09, 0.68, 0.23], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8fdf088dd8faa77add79414a359b508553b5764d4d7b7d74876c905e3f0e0bce:action", "state_id": "30a4ae3951edd9947b72447383f95521180f6ecb4f2d9a00939e2c224fb6dc87", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.52, 0.23, 0.25], "teacher_probs": [0.52, 0.23, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.625, -1.6953125, -1.91015625], "student_probs": [0.37257838249206543, 0.34728124737739563, 0.2801404595375061], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.52, 0.23, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cb52162210812188409f115a0133977fc7bf54c27e29f683dad8483844111042:action", "state_id": "66c26d98aaec17346d324d638883186d06dbdcba09a0745bac4e999f53b76d4a", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.45, 0.28, 0.27], "teacher_probs": [0.45, 0.28, 0.27], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.232421875, -2.015625, -1.265625], "student_probs": [0.4124932587146759, 0.18848468363285065, 0.39902207255363464], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.45, 0.28, 0.27], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "18e40d02c149fa59e1c8b7907531fb9de09ddcfd4e1633541a967e32812f68fa:action", "state_id": "1b086bfd4034563497fc45914a779e37271cb4f96d0b0722828459e05f8948fa", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.17, 0.57, 0.26], "teacher_probs": [0.17, 0.57, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.84375, -1.146484375, -0.9482421875], "student_probs": [0.18325647711753845, 0.368025541305542, 0.44871795177459717], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.17, 0.57, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0482a00bca2f776d470611c0a5e8ca91e3eeb1bc456dcdd95326eac2372ee58e:action", "state_id": "9a3e31c6cb31a5a2862ca3cc4e8f75cb0627314e7b9ea13d5278f05ac781f85e", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.41, 0.26, 0.33], "teacher_probs": [0.41, 0.26, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.279296875, -1.203125, -1.12890625], "student_probs": [0.30850523710250854, 0.3329228162765503, 0.35857197642326355], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.41, 0.26, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0c49763a4e2dc58639e85a2b1f34904b2f7acfcdfab4b6f216b5fbb617841d2c:action", "state_id": "5cde9d1aa1f5f541692e61ff56ad51bce826942166eb5ef85712f6520e2c9076", "family_id": "unified_maze", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.22, 0.34, 0.44], "teacher_probs": [0.22, 0.34, 0.44], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.0166015625, -0.384765625, -0.6181640625], "student_probs": [0.228803813457489, 0.43039390444755554, 0.34080225229263306], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.22, 0.34, 0.44], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bcd3eabb47cebed08824366cb7bc937a44f03731760cd9c76e33c7d64a77170e:action", "state_id": "5e49d71c4532ce29cf250fcae4f0876d4a2b9128f12e53a792ebf9637e1240ba", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.99, 0.01, 0.0], "teacher_probs": [0.99, 0.01, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.119140625, -0.23046875, -1.27734375], "student_probs": [0.8858202695846558, 0.08451294898986816, 0.029666831716895103], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.99, 0.01, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0ed5038ac53cc03ddfe1f215fff8b35f622a755b6dc2b90eff9f2640e9032413:action", "state_id": "aaca155b883956b556181ab034ada77c326e632afc3d8cc219ef1b9b44b30551", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.99, 0.01, 0.0], "teacher_probs": [0.99, 0.01, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [4.255859375, 0.05859375, -1.0], "student_probs": [0.9801486134529114, 0.01473813783377409, 0.0051132990047335625], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.99, 0.01, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8f81550fc012a478b0a6b6453073c2b2215ffede7f501c4ddff99f87b22e4e7e:action", "state_id": "d6c9d60782691ada6df7f360f28621cf458d42c69585b09b8625698e584da762", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.61, 0.36, 0.03], "teacher_probs": [0.61, 0.36, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.625, 0.2265625, -1.23046875], "student_probs": [0.766570508480072, 0.18932956457138062, 0.044099919497966766], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.61, 0.36, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "205ae6ccaf19332507a04cd7818ae607f7a58d1ae38334b0c42edbca1f91cd22:action", "state_id": "70a4a3d30a2c82aacce3279fea347a15d9aa3b741f3446e2e42b9e725bf0fd13", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.67, 0.12, 0.21], "teacher_probs": [0.67, 0.12, 0.21], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.6796875, 0.0703125, 1.26953125], "student_probs": [0.758906364440918, 0.05584072321653366, 0.18525294959545135], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.67, 0.12, 0.21], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bf0d23fe5ed4ed54edd03f3a09809c8b78995cb4bd1bbb29c2defcac3a8377a6:action", "state_id": "59c5de510fe704d6a2a9b2b00eb5ff7b4d75d7a1758564552b8c23fd7a223e63", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.22, 0.06, 0.72], "teacher_probs": [0.22, 0.06, 0.72], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.05078125, -0.810546875, 3.091796875], "student_probs": [0.04059877246618271, 0.01899113319814205, 0.9404100179672241], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.22, 0.06, 0.72], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "be5bfdc92c9f667b73c791258e569f2de55ee85be1a2b0978688f0819ee901c9:action", "state_id": "2ca7f792aebaf5a9b760e676e9eee1649546a0e4e9a9b983630bd0eb05d2f8b6", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.14, 0.57, 0.29], "teacher_probs": [0.14, 0.57, 0.29], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.140625, -0.8349609375, -0.8046875], "student_probs": [0.11773433536291122, 0.4344560205936432, 0.4478096067905426], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.14, 0.57, 0.29], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6fe90314819d0e70d11bf0735b837e3a873907f0d54cb760ca7e4b59ae2d05b3:action", "state_id": "1f256f018b494579f6a6dde9168c7be5fc5844441fee129cfd5b1cb5ad225f90", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.14, 0.59, 0.27], "teacher_probs": [0.14, 0.59, 0.27], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.19140625, -1.15625, -0.92578125], "student_probs": [0.13585379719734192, 0.38250264525413513, 0.48164352774620056], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.14, 0.59, 0.27], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "29f3307c7ea15a6e84e9d26a8a7afada54603dc8f6c290564d682625cb4d6c6b:action", "state_id": "2dfc1dff4d2bd92ea75e3515b8487149f7811345d4d84f671d1dcc7284a23cd4", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.14, 0.44, 0.42], "teacher_probs": [0.14, 0.44, 0.42], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.3046875, -1.7421875, -1.41796875], "student_probs": [0.1929679661989212, 0.3386693596839905, 0.4683626890182495], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.14, 0.44, 0.42], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b30ab2c1ef21bf6ee78a51f06a1cea43813a20d956221e8ad248d1814cd63ec4:action", "state_id": "5f829cd7f2743cd7973b28207353e87161ce8392ed8d9c38b72f2f7fa54b7132", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.17, 0.43, 0.4], "teacher_probs": [0.17, 0.43, 0.4], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.5078125, -1.47265625, -1.61328125], "student_probs": [0.15970014035701752, 0.4496431350708008, 0.3906567394733429], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.17, 0.43, 0.4], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "411eaca21f12fd0b9f9661eca1e6e96fdfed0e98cb697c9ca341ea6ed284841a:action", "state_id": "4c479b04b4ec21cda6593351a9d62d06da9d66f0b58a31b2c8baecd790271474", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.12, 0.53, 0.35], "teacher_probs": [0.12, 0.53, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.42578125, -1.2265625, -1.36328125], "student_probs": [0.13867472112178802, 0.46005672216415405, 0.4012686014175415], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.12, 0.53, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "60b92cf71ddd712a19d16d27127778a89c1e985a37a8bb30ef279637130fed16:action", "state_id": "96209e609cc68ecd91695b15883236002679663431a1347e3a6f6f0a4b5370eb", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.3, 0.19, 0.51], "teacher_probs": [0.3, 0.19, 0.51], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.07421875, -1.84375, -1.26953125], "student_probs": [0.22246010601520538, 0.2801195979118347, 0.4974202811717987], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.3, 0.19, 0.51], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3fd67160d72efcea6bd25d323b0c353c423a129e9cddc82efe70e6f75c1fa63b:action", "state_id": "809deff481e240f9e3de582ab3a04b7a1c2a9ca228950308d7e54dee9dad6115", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.53, 0.44, 0.03], "teacher_probs": [0.53, 0.44, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.7490234375, -1.4609375, -3.27734375], "student_probs": [0.6367411017417908, 0.31245145201683044, 0.05080743879079819], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.53, 0.44, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f591786eb54f740b0e17c660870c80caaa1d15ba9c39810938a8da197ed32af9:action", "state_id": "f35a714410c6336ac5e0ef96f8ef3686ea4ab8eabd676ffb07db650d413ea658", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.48, 0.48, 0.04], "teacher_probs": [0.48, 0.48, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.642578125, -0.26171875, -2.5], "student_probs": [0.38173529505729675, 0.5586855411529541, 0.059579141438007355], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.48, 0.48, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8d780e7f91fc80644e0bbd6334bda05725d79a2096bfb1176cdb8926daaa1677:action", "state_id": "30a0c431e42194f2654bebe694fc7921a623b116407a7b5a3efb7aa0632891ce", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.55, 0.4, 0.05], "teacher_probs": [0.55, 0.4, 0.05], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.646820068359375, -0.3125, -1.89453125], "student_probs": [0.37255722284317017, 0.5204588174819946, 0.10698402673006058], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.55, 0.4, 0.05], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4ccf2cc6f5d2881f04c43b4337f6daa5311aef69273a86a1981315d4c997e0b9:action", "state_id": "ea6858645a215f5b0c1cfba4f928dac26e9a853c870898c944ef049d6bf33915", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.45, 0.45, 0.1], "teacher_probs": [0.45, 0.45, 0.1], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.03125, -0.6640794277191162, -2.078125], "student_probs": [0.35782307386398315, 0.516569197177887, 0.1256077140569687], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.45, 0.45, 0.1], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8b2b5e903b720603723897960b980ad8542a35e6df0de92b7f11fbe14f3a1c60:action", "state_id": "d1fc871cee4f8a8641531fdf2a13954ff863f331335172857ffa2e5b0ed7594f", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.74, 0.24], "teacher_probs": [0.02, 0.74, 0.24], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-4.2265625, -0.5517578125, -1.81640625], "student_probs": [0.01938861794769764, 0.7647055387496948, 0.2159058302640915], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.02, 0.74, 0.24], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8941b280878f9949b41694681b9fddefaf87f2019a1551f1d4a992219e8f28c5:action", "state_id": "dea4fcde756c1e3d3759b3ffbef28dec4143225adad472fc77eeb5fe297da90c", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.54, 0.45], "teacher_probs": [0.01, 0.54, 0.45], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.64453125, -0.1015625, -1.30078125], "student_probs": [0.021744029596447945, 0.751677930355072, 0.2265779972076416], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.01, 0.54, 0.45], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "82ce2170e768f99a78d06f0fb80e6b05c50bb7b31c693b22c33674cda7da5dd2:action", "state_id": "785495e26d48e38fb973f30be9745ae16051b87d5d084bd4254801e20d7a353c", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.5, 0.49], "teacher_probs": [0.01, 0.5, 0.49], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.56640625, -0.271484375, -1.068359375], "student_probs": [0.02491651102900505, 0.6721305251121521, 0.30295297503471375], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.01, 0.5, 0.49], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8c90545ee5e23a800a79c7f4328533120f2490efe9df6575aff4e88ac1e89ffb:action", "state_id": "d6c59caba901bf44fa6f8cbe7635cd7be453dcf29a19f2b3c93a714fe7732c95", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.0, 0.02, 0.98], "teacher_probs": [0.0, 0.02, 0.98], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.88671875, -2.46484375, -0.3916015625], "student_probs": [0.06826883554458618, 0.10409753769636154, 0.8276336193084717], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.0, 0.02, 0.98], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "991c616b6bec2c0bc44102733de6957f71c06e7d3d126eab4adde41bb9f18aa9:action", "state_id": "e37282614cb59047448b07c8c793c63a3413886778a016aac61b2d92bbd6c31d", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.06, 0.93], "teacher_probs": [0.01, 0.06, 0.93], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.15234375, -1.484375, 0.19140625], "student_probs": [0.02887958660721779, 0.15310189127922058, 0.8180184960365295], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.01, 0.06, 0.93], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2052829183242cd69f961af75a5ed52b939b96437660f3e3f00f9361e1f4c846:action", "state_id": "ba18554f3cba0942edd871a8b32a5d16d0cebd5cce6991cceebdb1744786ebe1", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.09, 0.9], "teacher_probs": [0.01, 0.09, 0.9], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.0390625, -1.72265625, 0.95703125], "student_probs": [0.016916099935770035, 0.06309692561626434, 0.9199870228767395], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.01, 0.09, 0.9], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b483a7701fef1d53ed38158e1bc3ea36104162991e8f9b00853e1396f845551a:action", "state_id": "76d6040276a8d2f042d0ff684c21e174b6a7bf6a986e1b71ae6269fb0a170725", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.13, 0.85], "teacher_probs": [0.02, 0.13, 0.85], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.70703125, -0.02734375, 2.87109375], "student_probs": [0.009643609635531902, 0.051727164536714554, 0.9386292099952698], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.02, 0.13, 0.85], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "593ca15aff8901609efa1d7b47f04071f1005fd6ccebbe4c59ede991499babc6:action", "state_id": "30b20e7c4b08cd16d6fc729f53b3b5fd8a1c46cb424ee12f09b1153dc96730cd", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.21, 0.78], "teacher_probs": [0.01, 0.21, 0.78], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.16015625, 0.03515625, 2.7734375], "student_probs": [0.018052220344543457, 0.05965520069003105, 0.9222925901412964], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.01, 0.21, 0.78], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9fb9806cd55401e4a0ccc859bb0363f7fb7b50d502901e78a52888d4c94f6163:action", "state_id": "483c8a4ae08d59fcc5602052a2e49d6194eaa6284238079efdcd54d566145a0d", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.05, 0.94], "teacher_probs": [0.01, 0.05, 0.94], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.958984375, -0.05078125, 2.974609375], "student_probs": [0.018325047567486763, 0.04544360190629959, 0.9362313747406006], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.01, 0.05, 0.94], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3a806d58e05841126bb62afd0495a5648145ba3238495c89b863bb02a04ae267:action", "state_id": "bc66993404f932e6a729ef66f43a54bf3f4f6338cdea34c1a712cc5ab66261ae", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.7, 0.29], "teacher_probs": [0.01, 0.7, 0.29], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.1015625, -2.0625, -2.39453125], "student_probs": [0.17080797255039215, 0.4827999770641327, 0.34639203548431396], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.01, 0.7, 0.29], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "27c6c7005e89a021b188e0af21b8ce7d2efa4e080c0166c8221f61b0d34f3817:action", "state_id": "9d2189a707d39bb6c9c1fb31cca8ae4323f28d8ae6219d78ca9eeb5da2668817", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.32, 0.65, 0.03], "teacher_probs": [0.32, 0.65, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.26953125, -0.0703125, -2.578125], "student_probs": [0.43106237053871155, 0.526089608669281, 0.042848002165555954], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.32, 0.65, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e28f6aca5b18d1ec11f417170e5460546234662db462e3e618a829a48f40ceb6:action", "state_id": "0648914cf6631d8d77cb80050849dfc94d69fe16622513e1fd2b10d5bb28c6ad", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.4, 0.5599999999999999, 0.04], "teacher_probs": [0.4, 0.5599999999999999, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.220703125, 0.171875, -2.89453125], "student_probs": [0.392190545797348, 0.5807532072067261, 0.027056291699409485], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.4, 0.5599999999999999, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "69f5ede56f6d24ab29cb3515ae10d1b7868a2eb20f616674a16573954f15a185:action", "state_id": "2ace18bfc5b67eb66aebdbf8c90603142504b2e5c56f201232df5000a7d0151f", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.46, 0.53, 0.01], "teacher_probs": [0.46, 0.53, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.35546875, 0.6015625, -2.21484375], "student_probs": [0.4245327115058899, 0.5429856181144714, 0.03248169273138046], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.46, 0.53, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f3b3511cce16751d59fb6b34f43c371d2179f98250cbecb393a61b1b1d96d70e:action", "state_id": "04c86c095b6084409cc071b0dc5a4493adb307b049ff211e596cc8a3a91b1759", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.7, 0.3, 0.0], "teacher_probs": [0.7, 0.3, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.5390625, -0.4794921875, -2.30078125], "student_probs": [0.8662926554679871, 0.11508467048406601, 0.018622659146785736], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.7, 0.3, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1921ac5fe335c8337387ac693129bb45831060efa480d7ec340b9d827cbf20c6:action", "state_id": "cf06249ec7fdaf78e096b645b319e08e8adb789a80352ee9d69f6a6738e0a875", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.99, 0.01, 0.0], "teacher_probs": [0.99, 0.01, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [4.3125, 0.05078125, -1.56640625], "student_probs": [0.9833848476409912, 0.01386380847543478, 0.002751357154920697], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.99, 0.01, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b063d66a651dc457cf290aff79ec9778adf1a68ab3271d0a7a15390d8f401ffe:action", "state_id": "122f34ce75e40c078b3a1234159f2be097610ba14979f8187b31c22ff63ea13a", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.62, 0.32, 0.06], "teacher_probs": [0.62, 0.32, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.80859375, -0.130859375, -1.52734375], "student_probs": [0.6722412109375, 0.26273977756500244, 0.06501901149749756], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.62, 0.32, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "12ddbccf54b76f7d49bf91f1e6fac00720701ed852c0ce9069d6ce3111c71635:action", "state_id": "9e9aeb7728df397231acfb823459282357633ed2545c0d87aa9a26c78ffa5efb", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.48, 0.46, 0.06], "teacher_probs": [0.48, 0.46, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.48046875, -0.505859375, -1.71484375], "student_probs": [0.6737330555915833, 0.25126442313194275, 0.07500249892473221], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.48, 0.46, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "119449d390cf264b32e8f6d43188264d21476ffe3c1a34e19bd7217871c9b546:action", "state_id": "96f9ba9ffeed8183af402b55ac817c86521b626d8c4d81ae4b30416eaff802a0", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.41, 0.53, 0.06], "teacher_probs": [0.41, 0.53, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.01171875, -0.435546875, -2.21484375], "student_probs": [0.5723205208778381, 0.3659268617630005, 0.06175263971090317], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.41, 0.53, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3b71d0b5237869d6fc178b14ef9424b8dc2f47f14a3a8cee48c8bc363a619ec1:action", "state_id": "3eb4db67411242c33199938b7044b4d1bd5c9143bedad3584f25070d19e0c6a4", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.23, 0.73, 0.04], "teacher_probs": [0.23, 0.73, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8828125, 0.55859375, -1.7890625], "student_probs": [0.17759869992733002, 0.7506449222564697, 0.07175636291503906], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.23, 0.73, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "46cc956dde040fe3e823cfae44b578159178cf85e4931804a8e9db27f13b84a1:action", "state_id": "cf7f34c20de484eb420c1b8886a104dc2d738437d57ff9c34566d7ac0b503ebe", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.41, 0.58, 0.01], "teacher_probs": [0.41, 0.58, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.2265625, 0.966796875, -1.017578125], "student_probs": [0.29545456171035767, 0.619398832321167, 0.08514659106731415], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.41, 0.58, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "20851d44071ec6355b1b77503833e138360ef05cd68b45bc1e7676c7fb123fb4:action", "state_id": "55b179dce5c6a14b12d4c2244521efb6413e762326fca87f59c728459e89ddea", "family_id": "unified_snake", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.99, 0.01, 0.0], "teacher_probs": [0.99, 0.01, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [4.02734375, -0.216796875, -1.123046875], "student_probs": [0.9802526235580444, 0.014064722694456577, 0.005682661663740873], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.99, 0.01, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "14c6c493410f3030c2ae2a0f5f2ff75cfed92987ab58f87e0c25a0a4d3c8e965:action", "state_id": "4980694492a9162dc171038b722d64881d65d74fe9bcf0e86b7d0942d7bc6030", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.52734375, -6.1640625, -8.0703125, -4.81640625], "student_probs": [0.7365036010742188, 0.05273057147860527, 0.007837699726223946, 0.2029280811548233], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "48770c185c195f71b603a6a423629bdc7e33307f729a0d9f70f26a3f4362c790:action", "state_id": "47a30f0468ca708b398c940be97f481b28302cb7ac648890ba5d867b0792b995", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.46484375, -4.32421875, -6.5390625, -3.19921875], "student_probs": [0.605114221572876, 0.094258613884449, 0.010290266945958138, 0.2903369665145874], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ed15620cec9b0f96247615f5abde8b93deb8336c3c7179370b6f1fd19d2bc85b:action", "state_id": "776d6553f5015af663e3fb5aea2f1a212b45db16bb179120136f0151c9f0567e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.61328125, -2.3125, -4.3984375, -1.890625], "student_probs": [0.43168818950653076, 0.21453756093978882, 0.026643555611371994, 0.3271307051181793], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c1faf7ba78404222a7d691f2e15f0401907b51c71b80759b7c2e0f41ac8bb51c:action", "state_id": "9acf484e6f809fd21e3dfd0f536760617d44a5fa79bf397abeb8c49f98070984", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37890625, -2.22265625, -4.4453125, -1.78515625], "student_probs": [0.4666733145713806, 0.20071369409561157, 0.021741507574915886, 0.3108714520931244], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2a7bfd9a4aaa2a5ed7d876ea14e50bf57290789a3c7fcc03289deb9ab9bff4be:action", "state_id": "4387b48ef3e6f192a4cf312c7b770c13bed2cf34f6e8131eb110fa68d0ffc38f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.94140625, -1.984375, -3.69921875, -0.5971565246582031], "student_probs": [0.7824947237968445, 0.04195954278111458, 0.007552376016974449, 0.16799335181713104], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "604558480c8f9bb40b9955e0961204167c966156b3eac289a9224db375399c34:action", "state_id": "92bb08edb7203a22408bd923dfd67c5c386dd6ca4eff5cf96ee15dcac2b76cb6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3779296875, -1.234375, -3.33984375, -0.0078125], "student_probs": [0.341962993144989, 0.14522108435630798, 0.01768626645207405, 0.49512967467308044], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "015230ef0ecb5822c1c01b496ecb8180e4b8147e2348bb57c97b3e23eb157d16:action", "state_id": "9fea7bb8174c5e0b7ba9ffce95f855d54f38f73620f6a7896dfffb6489c9d4a4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0703125, -5.96484375, 3.75390625, -1.03125], "student_probs": [0.007902856916189194, 5.917199814575724e-05, 0.9838201999664307, 0.008217671886086464], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ac6a84b19ee42513c5a83d86916d63e4ba31932498b2371f71c5c8340778f56f:action", "state_id": "9909da88763933ee1ca75a2da0e5ab0a03b512288b489a505fe05763896222de", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37109375, -5.75, 3.09375, -1.24609375], "student_probs": [0.011229251511394978, 0.00014080431719776243, 0.9759055376052856, 0.012724407948553562], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "95e9d0e085156e8ffac0869bfc2f7b9acb4f4c26de6e3f7f4d4d123a009e8f45:action", "state_id": "52d02033cfad957412b0a2c936513512ddd7eb81b717042379f93c312ac1fcde", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.107421875, -2.20703125, 3.580078125, 0.453125], "student_probs": [0.023354120552539825, 0.0028609796427190304, 0.9328770637512207, 0.040907781571149826], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e01b79bc1cadb0f2ff043ced1da080118041a4d62ffb787263ed5fb0506171d2:action", "state_id": "04d02bdda8895358684dc8f1211313d65ba45b699f45f5405fb5877170fe1030", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.0625, -1.90625, -3.6875, 0.87890625], "student_probs": [0.26677650213241577, 0.0422101691365242, 0.007109353318810463, 0.6839039325714111], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "17859b67cc33c74cc18822dd10d7c20a579e9a7315b2b07509de71d1860e1545:action", "state_id": "c76d2b513499af094a817767d8962bfdb95f1e346ec763a128ababc703931028", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0625, -0.562713623046875, -2.6171875, 0.484375], "student_probs": [0.31963691115379333, 0.17105278372764587, 0.021922165527939796, 0.48738810420036316], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "35b39ff2d23f10de96c6dd17f9041115f1d48ae31118df751283b2705f116f2c:action", "state_id": "5ea6284a5c6d735c7d00b64896356812954aa9d7197ca96a47f9f5789415977e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3974609375, -0.94921875, -2.734375, -0.0390625], "student_probs": [0.3222067356109619, 0.185570627450943, 0.031133340671658516, 0.4610893130302429], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c932c972ff4a082fbdb01b6db69ffd9cc0c624448b3c03e8f8f16395e4265c45:action", "state_id": "001b85b0ba43c94011c2e37278f6e7f589364106898d4f7ad22cae4c00e75b48", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.923828125, -0.806640625, -4.4453125, -0.10546875], "student_probs": [0.226210355758667, 0.2543351352214813, 0.006685767322778702, 0.5127686262130737], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1ac8b66be8c30dff09505111e70e3278463361b79388830ed1eb05fde19139dd:action", "state_id": "6a2ec4820b1f50d2df057506edab239c0476276e71409017f31a6e0629e72c45", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.849609375, -1.05078125, -3.35546875, -0.279296875], "student_probs": [0.2726134955883026, 0.22293563187122345, 0.022246742621064186, 0.4822041094303131], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9bf7e5e405b24941c80dff83e5a05d9eea4d0b63c2dee0251b2105916b264fa5:action", "state_id": "259c50192653ffbc06d02948d23ade668d1d8c3d4a25d3435ed0af82844f0f8b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.9609375, -6.046875, -8.125, -4.33984375], "student_probs": [0.767306923866272, 0.03505609184503555, 0.004387784283608198, 0.19324921071529388], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7fd9b6755c0f737b4f49540d87fc6d8f4e813a3a42374e764f84f05e9cebde2c:action", "state_id": "482add8b5bd5ec4f8372edef9c89a9cc8d164a255378a35908ab2df0bae8b74b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.50390625, -4.61328125, -6.6953125, -3.47265625], "student_probs": [0.6596323847770691, 0.08002249151468277, 0.009976940229535103, 0.2503682076931], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "58eaaf757468681f519cb07a9a9ac9ecff56c796bb03d1b354ce0bb00f6d696b:action", "state_id": "66203c3cb291eb4ee635cd58ec304c4bd4911ceee90d046315e636b54bb4e92c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.19921875, -2.6328125, -4.765625, -0.607177734375], "student_probs": [0.6612287163734436, 0.03894181177020073, 0.00461474247276783, 0.29521483182907104], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fe933139446a46c16c1d7560348287ce2a2e99dffc3b93d6bad0893efb10af49:action", "state_id": "eec1c1bd30692746898d0e5d7fc7d118fd061a0f4f393a3d640b92a9483a9a4a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.9765625, -6.1796875, -8.1015625, -5.18359375], "student_probs": [0.7014074921607971, 0.07747567445039749, 0.011337196454405785, 0.20977966487407684], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2c7b351926d15c1d0dbf6e73c9eb894bbe4323acd573c8bce4825e92f3e8de08:action", "state_id": "0ce10e72a2765c9fba0a3f7c2c1d8fa92d917ac9113f494ebce7f9fc5c03ce38", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.75390625, -4.453125, -6.65625, -3.45703125], "student_probs": [0.5889084935188293, 0.1076679676771164, 0.011892727576196194, 0.2915308475494385], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "29505eb32c39285a15d682cb2970b71921f47368780085cf4bafddac45fd6e76:action", "state_id": "c973b74efc700cb157309d4732b48949aeb40012dda1e4e633f33c68230bafc4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.01171875, -2.28515625, -4.71875, -1.72265625], "student_probs": [0.7735833525657654, 0.07964632660150528, 0.006986656691879034, 0.13978365063667297], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "83e49989d976f9b02461a173cf7874dbf1cdc7f865ee4573cdc917285cc039c8:action", "state_id": "c7a8596a7d5e589fcb85c4121df6d9e30a358e9b71698624ee615fe0385a364a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4296875, -1.6875, -3.43359375, -0.2109375], "student_probs": [0.3878391981124878, 0.11025306582450867, 0.019234096631407738, 0.48267367482185364], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7275081655fdd07ff76ee09e70a050c46cdf81ffad63a6d4443a889b08a29d9e:action", "state_id": "94601da785d46614cd979db298c65b410b37c779bff0d6b9fefbb8f25546e416", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.134765625, -1.23046875, -3.06640625, 0.4921875], "student_probs": [0.30679434537887573, 0.10256272554397583, 0.016355056315660477, 0.5742878913879395], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9f58c679170c4f62cecd89a8027c50d3f8d0358c0b5a098dcd577ce7d2bdbeb6:action", "state_id": "f033a9f2915f888e76c8701b5537c564bd3e6e589562b5787e5033083adb9823", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.52490234375, -0.9765625, -2.36328125, 0.015625], "student_probs": [0.2846928536891937, 0.18122705817222595, 0.045287538319826126, 0.4887925684452057], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f1b3b52237016b9c3d3bb1e6172031bcd0a67c99253e0df91283dd389781e58c:action", "state_id": "aeb981a2ec7f2ebc238e0661a3e15d9855a9778b43fe3188582ec783a5fe8e2e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6595458984375, -1.484375, 1.201171875, 0.2109375], "student_probs": [0.0975160300731659, 0.04274224117398262, 0.6268670558929443, 0.23287460207939148], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a1b7be41ed56ac45e02cec73a3feb4dacec015fa52c4f3f43f2915362a3be8fc:action", "state_id": "7ad574021223b85cb3fa4062605d8f96a3146e5ba0b4c956776cc469efb5a75f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.71435546875, -1.61328125, -3.38671875, -0.166015625], "student_probs": [0.31186914443969727, 0.1269327998161316, 0.021546650677919388, 0.5396514534950256], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "feeaff7bd5f569f4773ea2bdb1051b139a02199089eef6e80257dd37112e727e:action", "state_id": "c9d855efa8fe94325aee3157eb81e3df58e117b0cf3dbc254795b1c25b7ce36e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7392578125, -1.41796875, -3.359375, -0.08203125], "student_probs": [0.2849409878253937, 0.1445421576499939, 0.020742088556289673, 0.5497747659683228], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fba0370bbdf5d4c7855de185df695d3e830fd514510a068f5df39fd5855af8d1:action", "state_id": "858b1ef83692226c23474d6742be7fbc2cb5c2dceba5468c2b1bdcfa62a9c01a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.66796875, -1.6796875, -3.64453125, -0.3515625], "student_probs": [0.35884109139442444, 0.13047228753566742, 0.018289316445589066, 0.492397278547287], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "236b1264f64401d8fe3985940988eefe1d63cde46d9acbe11da387f65fa52b98:action", "state_id": "af4d12f49931596eeca9048146ad776b4f36d97a89a3c3f210333e5add8ec691", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.251953125, -1.6796875, -3.265625, -0.26953125], "student_probs": [0.4402303099632263, 0.10559000074863434, 0.02162015810608864, 0.4325595200061798], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dffde3c4d5dbd1988d36d1c68d8d4878c6bb38309ae7ca437b2da727ed6cde96:action", "state_id": "f88068ff21ecea4a4795770c3c5f80aaf4714ca61113dc93fc817f555422dc11", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.47412109375, -1.009765625, -3.4765625, 0.1875], "student_probs": [0.27988556027412415, 0.1638147532939911, 0.013900702819228172, 0.542398989200592], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6108ce55bb3ebe7113b612e7b75dedf732b3bd9b226e2ef5ccab1efca12dc129:action", "state_id": "55359aa9068fc6be27ebe68b8985b3fc93c1b2231a689c1a0a591e08da75abe7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.55810546875, -4.5234375, 2.7421875, 0.265625], "student_probs": [0.032874695956707, 0.0006233613821677864, 0.8915809392929077, 0.07492095977067947], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2ac5e1ae273ee00229178636a2ec2f773558fb8301596297529eae7d60f8f9f3:action", "state_id": "9ff76f9f3f4c78af3ec85d240f10092ed41566db6f2dbaf4fc0ed8635af5df06", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.376953125, -3.26171875, 2.96875, 0.15234375], "student_probs": [0.03211909532546997, 0.0017944256542250514, 0.9115567207336426, 0.054529812186956406], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5c115a0aabf7c8ef42e96f8c542d22526dc4d4450ee9d95b80b3175a78036791:action", "state_id": "763d6ecfa0902f9a9a2af37fd1cf5594e7b55124100d4b98faf5d6138fe76936", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.341796875, -2.42578125, 3.533203125, 0.328125], "student_probs": [0.01950792595744133, 0.002427438274025917, 0.9399445056915283, 0.03812013939023018], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3bd4de08e47c956a02671a5eec274ed583773810cb8feeb117d114ccdbebeac2:action", "state_id": "9ee568fc5035cfda2cda953d2aee42183eabe500def7a4ab1199e4cef56b093b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.73046875, -2.19921875, 3.0546875, 0.64453125], "student_probs": [0.08203607052564621, 0.004381852224469185, 0.8383014798164368, 0.07528053224086761], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e3c25e389b1132f88996bfe1032ecc833af6ea3e7e0b0a88bab55dee6a7d7849:action", "state_id": "ea7e02fedb4cdbab885dc78efdd06f9042a54a62f05f646d96023d64f957d14d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.03515625, -0.6416015625, -2.78515625, 0.48828125], "student_probs": [0.3183627724647522, 0.16181178390979767, 0.01897038333117962, 0.5008549690246582], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "39ed814e5dbcfe54cfdeb741b562b622e054eae39e4ee5c46727f113a4c41820:action", "state_id": "cfd35867d00361c47a14c3bd81b8068985976a48b4bac35e6e77e90a4ce7c00e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.849609375, -0.46435546875, -3.2265625, -0.123046875], "student_probs": [0.21594631671905518, 0.31743839383125305, 0.020046943798661232, 0.4465683102607727], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "22db6d260d3fcfd15dfb78ace7f9d5777a6b502831b7ea0047ee18bdcde0568f:action", "state_id": "5bc225df501aaf9d78c67724d4a2caf2d4fd964681b51b870c8228e67c0545d1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.39453125, -5.4765625, -7.6875, -4.54296875], "student_probs": [0.6870619654655457, 0.08566062152385712, 0.009388220496475697, 0.21788926422595978], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c104c9622a1d5ca89a4337206ec9c78c3c34928543b01f423d856f6a17184149:action", "state_id": "6b56af3879d565bb665eea133c20bc36157e9f56c4545eb0daf776495fb06b6f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.109375, -4.03125, -6.3515625, -3.375], "student_probs": [0.4538075029850006, 0.18051216006278992, 0.01773403398692608, 0.34794628620147705], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "07b68a43163bb4e2daf19b2fa41310877a386ede7835676aba6844d37f372100:action", "state_id": "35aee78e1f285fa5710b3e1180902649d47ecad80de252fcd517a30a9bc0f58b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55078125, -2.30078125, -4.875, -1.65625], "student_probs": [0.41523584723472595, 0.1961435228586197, 0.014948752708733082, 0.3736717998981476], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b583ebcb6c198f0727cb1dc61ba9283f8ca36e8ff259901fc2aa83403c7f00cc:action", "state_id": "12ea8cb8677c6deeae6f03583d19dc433c0e8e0a191524a1159cfee6550340a3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.74609375, -1.62109375, -3.99609375, -1.3046875], "student_probs": [0.8122830390930176, 0.0761466696858406, 0.0070827435702085495, 0.10448750853538513], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "168b7fcf084a16203ed8ce80e9cc9ec43bb6cb4e4c1dd5ad68bdd6714ecb5f86:action", "state_id": "4a38502af092638cd69098611d74f50883134cd8e6240e4dc1b73e293adcbf49", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.1875, -1.0390625, -2.93359375, -0.12109375], "student_probs": [0.39068812131881714, 0.16672521829605103, 0.02507360652089119, 0.41751304268836975], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e6ce502e6068e3714aec5f2aa66a61e934aa6a1388907869c7bb0d78f68aec67:action", "state_id": "55502c0f52faaf144483294e2eafd6487c0e4a2b57f0d8fb32d12043d872d847", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.345703125, -0.95703125, -2.9609375, 0.37890625], "student_probs": [0.2717609107494354, 0.1474655568599701, 0.019879484549164772, 0.5608940720558167], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "31e57c062b15fc40bac0bd17867aa359cb995d71c70265a15652d3234f00350e:action", "state_id": "fc876da4c0416c2311ca4eee32d1ee3b51b711293f235ad3d4ba1c080d038d23", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.25390625, -6.1484375, -8.09375, -4.02734375], "student_probs": [0.8382154703140259, 0.01706012710928917, 0.0024386178702116013, 0.1422857940196991], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fb973c6515b87c73776201fc2bbe3a40cba17c52dcf4c875293a6e73c585c4c9:action", "state_id": "3f3bffaeb04ea64f1a8a599727e06727c1b24df6d2224549a0f73061e96d712a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.71484375, -4.36328125, -6.703125, -2.625], "student_probs": [0.959440290927887, 0.005978831090033054, 0.0005760167259722948, 0.03400495648384094], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b3ed36bd896dc638bc3af37017d0b83efda139388406f2e328c9d686e030c49c:action", "state_id": "e77d9429904b8f3e1a0a45508c477580dda0dcbdd4c2d9a2e722ac0f2dc48c75", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.613006591796875, -2.3671875, -4.26171875, 0.015625], "student_probs": [0.3252967298030853, 0.05629224702715874, 0.008465724997222424, 0.6099453568458557], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0b515dd76d6cc68ae9d1648af61fc69989431d91c62c76f33b2c6ed1f7ddc4a9:action", "state_id": "372dcf0f68ea707817a737fe252d6d2588690335e15fa3ddfd6be63bd58a9714", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.40234375, -4.7265625, -6.828125, -0.19921875], "student_probs": [0.22878527641296387, 0.0082364147529006, 0.0010070271091535687, 0.7619712948799133], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d7ac191ad616792f430d07b0d2cfd8c2aae23083fda9999d946080551fe51ca1:action", "state_id": "faf3ac8ceca09356ea43eece6006bdd013772554bf27ebf3d39aac3f62d11500", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3203125, -4.8203125, 2.83203125, -0.08984375], "student_probs": [0.038969460874795914, 0.0004329115618020296, 0.911527693271637, 0.04906995967030525], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6fee8fa52ccf53d7c95c99d24a5e5eb21f3877e2677b1e34b8778f8015654f76:action", "state_id": "a2ce9267ce3dc92a9f90bd4cb2f0e77f47cd84d8f502b8cd12b1a9a51d99a176", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.359375, -2.9296875, 1.75390625, -0.26953125], "student_probs": [0.09573166072368622, 0.0073245856910943985, 0.7922130227088928, 0.1047307550907135], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "332059d2a0c2dac4b5845fec088c50e76d7549c12ce0bdd99c1b348c94751aaa:action", "state_id": "6478b4b7f8f11aeee07a32a7f2cc212488035dffc65085ca48ad1d19c299e17f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4296875, -0.98046875, -3.43359375, 0.37890625], "student_probs": [0.25834178924560547, 0.14893384277820587, 0.012811936438083649, 0.5799124836921692], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e885ac73c1793d41cdf83524d83ec22cb67ee7b8c50db89e7f5b2892f466fbfa:action", "state_id": "0db0095729fb1db46a081ea157cdc162fd4460ed219ce95c63969473a152e0df", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.287109375, -4.765625, 2.80078125, -0.0703125], "student_probs": [0.04134929180145264, 0.0004693247319664806, 0.9068217277526855, 0.05135961249470711], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "19df245a73682ca6e44e9857d9a615b8a101ee93107426fe6a289939507acd6c:action", "state_id": "5434901a784daf45aa485192d12469fb006a9654d6cb237a18528f0cf801f74e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.32421875, -2.9765625, 1.515625, -0.298828125], "student_probs": [0.11916455626487732, 0.008399412035942078, 0.7502070069313049, 0.1222289651632309], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1c03777e7cf2879e87be13718ecea46674c3245d360fd7040f76f787c05253df:action", "state_id": "c3849d15547f6d33cb15fcbe9e9915cbcb013eacb69f746524fd05339da66e20", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.26171875, -1.62890625, 1.24609375, 0.0625], "student_probs": [0.13976998627185822, 0.03561655431985855, 0.6313185691833496, 0.19329486787319183], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4138de3504233f826014ac6b9d82c033ef7b3a131af3eee4c5a0cf2a50bdae87:action", "state_id": "450bf684a3ac5d4a3fef63baa879f2c67e87dc347c869a023b316e1f263ddff0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.349609375, -1.34375, 1.890625, 0.55078125], "student_probs": [0.07560785859823227, 0.02797803096473217, 0.7103761434555054, 0.18603798747062683], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cf768ad07736a636d4ab91385f7d8042e7b24be13f54face4c0549628f42979e:action", "state_id": "af9f164e93aadb483a27b7293980344a64c136c08878c2bbf0662e66b0977ca0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.2421875, -1.087890625, 0.0703125, 0.65234375], "student_probs": [0.19075660407543182, 0.08188331127166748, 0.26073339581489563, 0.46662670373916626], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2b40be77f8b0b95925f4189930000bd6e0648ca587e1129e1ea6a63fa62c2e7f:action", "state_id": "d4e3e50b3b071e84453b3b5efb5b116ff8ee5b94f7af75944a87b0f1869ef13e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.11328125, -1.044921875, -2.7109375, 0.45703125], "student_probs": [0.30891138315200806, 0.12168233096599579, 0.022997790947556496, 0.5464085340499878], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9dac5660047a099a376275dbcc35a932bcf7d1ac7985de2744acb3c2c569073c:action", "state_id": "61b349e6ca6d26e099635f7b218970c599750d852819c015617434f4f68eee40", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -4.546875, -6.0234375, 0.14453125], "student_probs": [0.23414823412895203, 0.006947461050003767, 0.001586949685588479, 0.7573174238204956], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7a18e8bad47280acee12c10c10ba859fd6e6f354900f4dff50a78afcf17217a8:action", "state_id": "f4a5e30c21b4bd3cdedd30f4e6bd087dc5ccc76056308d705be08d127ad7bd9d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1640625, -6.046875, 3.625, -1.0859375], "student_probs": [0.008178121410310268, 6.195480818860233e-05, 0.9829172492027283, 0.008842657320201397], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c8ea81ef409dac433b46d5bf3eb33e6b420ec3ed48e4f6d6c7ae92518ad19e27:action", "state_id": "0e6a0a1ece0dd2d1818dcdd89ec51985c86d81316c6c9b9d1044a7c062837d8b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.359375, -5.7421875, 3.109375, -1.203125], "student_probs": [0.011182030662894249, 0.0001396655716234818, 0.9756052494049072, 0.013073117472231388], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b6b254fcc292600a81207112ef27afe33813ba8afd00e23c5701559c0b57a89b:action", "state_id": "b3784838cf79619aae6de26f8b86cafaa96ad01615d0d20d1bdb11e34a3399bf", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.142578125, -2.2890625, 3.236328125, 0.30859375], "student_probs": [0.031224912032485008, 0.0036500173155218363, 0.9160972237586975, 0.049027830362319946], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4b443955e2f98be977be7d9dc362deb2fcf94a8e7ef75fbcf1cc191a30deba38:action", "state_id": "be69cfddc7e68ca606449fc1c5cf382f27bdd1fed109029793dfb2577eceea5c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.12890625, -0.505615234375, -2.6015625, 1.125], "student_probs": [0.23239262402057648, 0.1232120469212532, 0.015149379149079323, 0.6292458772659302], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "023a86b6c23d71241ac7cde7c382b9208f40dab766e0aafc25751cbc46a704c0:action", "state_id": "d42de08d180dc2d76443885db2084e1d671943e7a0c91c44b24563350c347403", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4765625, -6.55078125, 3.515625, -1.3203125], "student_probs": [0.006691936869174242, 4.1864568629534915e-05, 0.9854425191879272, 0.00782366655766964], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8e1f822b73a25b46f37e6b0b9440192863b8f16674fb1bbc80c96345a0606456:action", "state_id": "671d310a2b3cc3ea9ca9b11b568761adb3f31a0ddbe7ba7a027a246527cb02e0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.548828125, -6.2734375, 2.984375, -1.533203125], "student_probs": [0.01051737368106842, 9.333305933978409e-05, 0.9787063002586365, 0.010682997293770313], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "48cce345d55dc0b153465f1084cbec7fecc8fdf3b15d9c170c6e918c656529fa:action", "state_id": "a44ab2014ddad9ab985678efbd3e856c7d95ad8b81b461458057496ceff7b2b3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.109375, -2.2109375, 3.078125, 0.33203125], "student_probs": [0.03716789186000824, 0.004544341471046209, 0.9004956483840942, 0.057792067527770996], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0d52acf512948929faeaff64cf49299998a9678f07889b7394b48764582d97a1:action", "state_id": "a3c7c368402a0299d15fc7b20a21ee7dc3b47a40445d112f8bd0afd7ae28e2e8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.16015625, -0.541015625, -2.42578125, 1.09765625], "student_probs": [0.24242901802062988, 0.12024569511413574, 0.018261071294546127, 0.619064211845398], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1ca94b9aa2e623a4f504dacb6cd40dfbe69aede6ca5d11b5c631c8e558d1785c:action", "state_id": "405daf63275ccb64104c0a45a36c87ef5d7b717f4934a0e1e75d56ba0bc8eb0c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4228515625, -1.07421875, -2.58984375, 0.1953125], "student_probs": [0.2864203155040741, 0.14932022988796234, 0.032801300287246704, 0.5314581394195557], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "905da7b3208f99e08f67d9d51f629b6cc478648b2ae5d2cb223621364da751aa:action", "state_id": "3ff48a7a9c91dbde7bbd940a565c5378b24c60026825191f47123bba0b09903f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.03515625, -0.7373046875, -3.1796875, -0.3759765625], "student_probs": [0.44448524713516235, 0.2202511429786682, 0.019151588901877403, 0.3161119818687439], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b3755a947a0cb630ae10c678b1621c93cd754964aeb13c1585b0bbed05e7e14a:action", "state_id": "6147da6015b29fcd6f0d8e9db106a7d00f5a956dbcc163596b505c37bab8e7ba", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37109375, -1.33203125, -3.71484375, -0.2421875], "student_probs": [0.1912750005722046, 0.19889451563358307, 0.01835610531270504, 0.5914744138717651], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cd65689db531137a6ac36c91d5941512bf25bf6f54fd2d69fb993ea4b9c9abd2:action", "state_id": "6c73127d3ede89a57d041092e156bc25ed09f3be728f25242d1cd47f4d9e1c1f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4453125, -6.5703125, 3.515625, -1.30859375], "student_probs": [0.0069022648967802525, 4.104236722923815e-05, 0.98514324426651, 0.007913485169410706], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "04e41fbfe7901f9f79b169cb637b5e0beb9f22e2a179c65c53f75de236079862:action", "state_id": "1d59be8a92497e246c79844de6c260d4370f0bbbef6f068a391d85c02f73ed00", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.564453125, -6.328125, 3.0, -1.560546875], "student_probs": [0.01020173728466034, 8.706381777301431e-05, 0.9794695377349854, 0.010241665877401829], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2ff108bc29a38d44d31840ca76a766b92ad179a0b81f5a56aa1fa2205d8a2c34:action", "state_id": "bef8cb57c5b8f640368fc868a126b8a924d79e2c7a9619b26bb219f25938190b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.201171875, -2.203125, 3.22265625, 0.30078125], "student_probs": [0.02987421676516533, 0.0040351469069719315, 0.9167400598526001, 0.0493505522608757], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b826b81e285c5985630bedaaa780b4e55c0844b0a00d92203b0117009d3bfeea:action", "state_id": "ae0107bb8d24ea604aebe02ef5f8f808c25a3d73f86906918a48fc1517f7c7cd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.140625, -0.5323486328125, -2.5, 1.0703125], "student_probs": [0.24300019443035126, 0.12397606670856476, 0.01732996664941311, 0.6156937479972839], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "02e46a99365d3695d4586c69024f7e3134500a44b742e435719361bbbab2a6de:action", "state_id": "9d181b51febf3bde7f2ab3996f1ba4822cca54c829918eb40d71ea317d095ce5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.28515625, -4.7421875, 2.80078125, -0.078125], "student_probs": [0.041442885994911194, 0.0004806023498531431, 0.9071009159088135, 0.05097562074661255], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1f90ddd50a754c0ce608f6cc826c69ceccf7a6c7fe8c6908dc19639ed3d4d128:action", "state_id": "c318704c0a58141aaa66fa49d62cb26d2a600a5a51e09997a6fbfb3f10295aae", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3203125, -2.96484375, 1.453125, -0.294921875], "student_probs": [0.12519055604934692, 0.008893366903066635, 0.7375062108039856, 0.12840992212295532], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3f771a4e9faf441746360df57b526519a2c2ee6f3f1f92ec9f0dd76e9b1a24ff:action", "state_id": "582ef215e923dee684b1e6cc976dbd6e126508ded206312c38be9e31702e4e37", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.353515625, -0.990234375, -3.23046875, 0.390625], "student_probs": [0.27100539207458496, 0.14336875081062317, 0.015259246341884136, 0.570366621017456], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bae55c836b06b34e49f941e1766dd789f58abdbc6e854e45a680af572e7d9a3e:action", "state_id": "91804804471115543b96e5a181fab140f36f0403e8c9e2468447c41250c6920f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.546875, -4.88671875, -7.078125, -0.615234375], "student_probs": [0.2794799506664276, 0.009905466809868813, 0.001107029733248055, 0.7095075845718384], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5a8e1ce1bbe6a8dfa7fc62bb16ac6681f519486d8c6edb1354e0c460993245d1:action", "state_id": "8904f24da227a2efd364b036bf1c2df2a39933a1fa15240ce240eff1e4d9ef41", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0234375, -4.546875, -6.0703125, 0.1328125], "student_probs": [0.23730868101119995, 0.0070000989362597466, 0.001525751082226634, 0.7541654706001282], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b9d0724b17b7610ebd7b08e19cf7edcac37c49b79544e7ae699b20adcdaa4900:action", "state_id": "50a1baed2c99effa427264b29f58837e321c6121d02912bd3d6f5923f4ff5d93", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58203125, -6.578125, 3.484375, -1.35546875], "student_probs": [0.006216417066752911, 4.2049825424328446e-05, 0.985944390296936, 0.007797133643180132], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7c9af825040dd965e22369359fb15dafd7474f24dedfd22a0804be069eb69948:action", "state_id": "2a17d66a5af266c47f2d617f003f345aacd4edda218eda4a3e13f26b7ea4267e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.556640625, -6.2734375, 2.96875, -1.498046875], "student_probs": [0.010593078099191189, 9.474216494709253e-05, 0.9780799150466919, 0.01123231090605259], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3fab73a847a667efc7ffae45535a75817ea2380283b8cbd0e350ef9e9060f3e4:action", "state_id": "ca6f3af6e0b8fd7017ec1441b14a7b8d13309acb95f690ac795990cb94e92165", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.08984375, -2.171875, 2.970703125, 0.33984375], "student_probs": [0.04166548326611519, 0.005194715689867735, 0.8891091346740723, 0.06403056532144547], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "76253c205d3a18fef91bd0ae2c84c060f96c0c2ecc5952ba049645533d7df3ca:action", "state_id": "abd56264ee7394b85a8ec3d9e43090568d5cb954e18dc9d18ad840b7769d2439", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.9140625, -5.9375, -8.125, -4.32421875], "student_probs": [0.770300030708313, 0.03746258094906807, 0.004203184973448515, 0.18803420662879944], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1c0541618cf6c38f8be4d0d2a5405a527062792d23dea5be6c0e270fa5001b43:action", "state_id": "4d039e4c5ddf58a281252aca981d79e574110f3123f14f182c6296110fd31378", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.421875, -4.3828125, -6.53125, -3.296875], "student_probs": [0.6353214383125305, 0.08940652012825012, 0.01043072808533907, 0.26484137773513794], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dcda8b189e3aeb3749429022e57c05768d4802e1dc7a709b4158d2e888962b81:action", "state_id": "96d9e104d30c80cee1b89abfc077bdf458d222d6aba752d1b72ddff95762530d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0625, -2.71875, -4.71875, -0.6566162109375], "student_probs": [0.6420425176620483, 0.0397816002368927, 0.005383854266256094, 0.31279194355010986], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5b7de3bbcaf360bc5f345c0aff448f90741d9145801df3e3350e3805b9703b56:action", "state_id": "3100945c88ff0d67eee0ba9fae0b6917399a55b0ddb4f34ee6eed52e3650b49b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.015625, -1.67578125, -3.40625, 0.5625], "student_probs": [0.3326137959957123, 0.06323296576738358, 0.01120496541261673, 0.5929481983184814], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f4331d6109eccf409df2281e5a0f0c9faa36151e2abf8e8ebb385d0600958530:action", "state_id": "83b7714411c869737c90fb801a28ad00bd90cf186583c1d6ff755c506cfc74da", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.296875, -5.671875, -7.921875, -4.4765625], "student_probs": [0.7091228365898132, 0.06595870107412338, 0.0069519951939582825, 0.2179664671421051], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4ae0a84ae0c41a7b020fdbd2e93d559efa713c187f17e6a6d8e993ded85136c5:action", "state_id": "ffd90fd81ad629340330f74ba9bb1179e793e5ead9dba445218f6b1f9868add6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.875, -4.14453125, -6.3515625, -3.21875], "student_probs": [0.49480873346328735, 0.139023095369339, 0.015296267345547676, 0.3508719503879547], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3a776d84fee6c59a9faa1691af5df30a28a6fa8040a9302b7e70d496134ad6d1:action", "state_id": "069595920523da87a806bd11c3d9fd2077e82e67e0c9581e9bb5ae7a5b438def", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.82421875, -2.67578125, -4.7890625, -1.96875], "student_probs": [0.42666780948638916, 0.18207947909832, 0.022002629935741425, 0.36925020813941956], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8fc481f842dc01d38926ce8f5aa6fcb2c5ec2feb7e05aa99a34949007d0c5ca3:action", "state_id": "156846720dca1592e3e4cef2132c16ede1571e4302c66c1cfc286496e5d2addc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.50390625, -1.703125, -3.78515625, -0.26953125], "student_probs": [0.6308476328849792, 0.06941015273332596, 0.008653828874230385, 0.29108837246894836], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d6bf78673ba85099360b277c0349fc6304ac548374197f29258345928ffa1e41:action", "state_id": "01f32af6f417d83fc3e72f4fe3009dd4d36f18dff15e383edc8f10f2d6ce96c0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.1171875, -1.34375, -3.12890625, 0.3125], "student_probs": [0.3473086357116699, 0.10186529904603958, 0.017090026289224625, 0.533735990524292], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2416cb71720487b0a470fecddbac5b6f756c0774a53e40395fc26b7bdb84f4de:action", "state_id": "ca613f2e3b54deee10f7f11d8826d5ba5128f83909166fee7eff9213302b6571", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.69677734375, -1.22265625, -3.19921875, 0.34765625], "student_probs": [0.22149822115898132, 0.1309133619070053, 0.01813734695315361, 0.6294510960578918], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ec82e002d4c6156480b0ba63f6bd0c6684ebd23256bcdfc88c4fc09cedba06e1:action", "state_id": "36f03cf613668f5c213f31ea0e614bdb4b7e1dcb8ac089dccf2992702d0b64b7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.73046875, -1.38671875, -3.88671875, -0.07421875], "student_probs": [0.2866209149360657, 0.14869697391986847, 0.012205791659653187, 0.5524762868881226], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "11042b52a916edcf21f46de906d233e2c773e849d6ba8bfd3866d8ebf5eea2a8:action", "state_id": "f1275aa3d3fcf011558b14f165ac436bcf018a44608bf65382615d6ae8a93a72", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.1328125, -5.6015625, -7.875, -4.3046875], "student_probs": [0.7126588821411133, 0.06035555154085159, 0.006214065942913294, 0.2207714170217514], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1fd146fe6a5d156f07ea47ed560db8cc395774e190f6cc1f3b6a62fb42f13ead:action", "state_id": "1ea40a6e44124aa19d272df558d4d1e8a5fe478049cb0347cf1edb4dcf9bfe4d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.87890625, -4.05078125, -6.3203125, -3.2734375], "student_probs": [0.4960806667804718, 0.15367862582206726, 0.015884317457675934, 0.3343563377857208], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ec0ac7aaa9a33b0d5f8c8e9e70ba1c046cad245391ac43654bf2e261ad020930:action", "state_id": "4c40e3fd4dcddd530cf4550921da0a57afcd6ab559c5fd4169638e4164a042d0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.921875, -2.796875, -4.6953125, -2.03515625], "student_probs": [0.42154809832572937, 0.17572738230228424, 0.026324402540922165, 0.37640008330345154], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "086d56bcdba61d9a7ec3821eb6a27135a6d9470132bdcdc7a43f82e3e03cffb1:action", "state_id": "28accd8d284cabefe3800be73077a71bbac4411caeba58208e3568c489224548", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.34765625, -1.5703125, -3.84765625, -0.419921875], "student_probs": [0.6149657368659973, 0.09034158289432526, 0.009265094064176083, 0.28542760014533997], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bb9fa1adc40f1ba2124dae619214ca11b4e5837b684b05a9232a21eb55582243:action", "state_id": "7de607c1c7bab59246852d8c2327075f87b5936bddfcb286ad09d5cdbaff2755", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0078125, -1.02734375, -3.06640625, 0.37109375], "student_probs": [0.3521825075149536, 0.12508496642112732, 0.01627989299595356, 0.5064526200294495], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a3bf0fe9bb33b8395464546b826f4320ef6fd6cc7c0d70fdba8a95d7d2024c06:action", "state_id": "ee16b6d3d3ad9bf30832c4fd05e07517c9f46ea1d6df63961b7ba0e94938d7fe", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.56671142578125, -1.01171875, -2.96484375, 0.23046875], "student_probs": [0.2531016767024994, 0.16219250857830048, 0.02300379052758217, 0.5617020130157471], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8634a55b8085fa0eb44ab33817bd5420d1c1c49b0765d084346d61236692ce6b:action", "state_id": "eb6f3b43e22ed1ab9f05b259ae8ac644fc44101264e2601e63bc2e116d0c0b71", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3916015625, -1.35546875, 1.162109375, 0.21484375], "student_probs": [0.12587636709213257, 0.048011139035224915, 0.595267653465271, 0.23084478080272675], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e2feecf029813391ae39238b9ff1b0082889b3d30e98fba2f521c3bcc0ba877a:action", "state_id": "3ea6da6da37a78ab3704b61c4806127529b9db7803daffed45764dd261ce0c82", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.568115234375, -2.05859375, 0.92578125, -0.4833984375], "student_probs": [0.14775212109088898, 0.033283356577157974, 0.658149778842926, 0.16081471741199493], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "198bfb77f7db807b8b55bd70aa30da9aeef4f1032aba0a0894ced7a2bfdd19cc:action", "state_id": "4506251c7346109d9a2d234854a7976948812219b1110cba03e9b18112b33159", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6134033203125, -2.01953125, 1.29296875, -0.443359375], "student_probs": [0.10918127000331879, 0.02675928734242916, 0.7346407175064087, 0.1294187754392624], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1933a3035d8c760db6da12d5ab9d69fa3d3e20ada1c309ae39f85ca6c926b5a6:action", "state_id": "8367c59e213c0b2ecd080976d592edb924f484dee3dcae24116a0b56156d7ff7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.990234375, -1.96875, -4.265625, -0.447265625], "student_probs": [0.3190016746520996, 0.11990272998809814, 0.012058933265507221, 0.5490366220474243], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e692a93f7d8796782258fba9c787dfaa3e40e30bdb95c5d0934d6c6d8dee6612:action", "state_id": "eb9fad4a08c53e0226724a62c41bff81144acdc458b59204586b65e06239f156", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.9140625, -5.9375, -8.125, -4.32421875], "student_probs": [0.770300030708313, 0.03746258094906807, 0.004203184973448515, 0.18803420662879944], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0c8d186b638e96ec712ca1597333a182048379695118f96dc4326a12c224926b:action", "state_id": "718746f6438ff3c5fb2bc0825f478ae29c5071da2a5cb19a680864a88b240712", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.7265625, -3.68359375, -5.5625, -2.984375], "student_probs": [0.45137861371040344, 0.1733435094356537, 0.0264794509857893, 0.34879836440086365], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "68ea212c456c45020af516286a4ec0819565432b10f46220332a8794578ce4e6:action", "state_id": "b955f0b74ebd218fd99b76ed7090a2d0d8cc052fa8cd9bdc9477a2493df1f041", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.90234375, -1.92578125, -4.3671875, -1.83203125], "student_probs": [0.3190097510814667, 0.3116198778152466, 0.02712288685142994, 0.34224748611450195], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a4a24a916b6dd27206aa98c20365ce426487df0c4b739fc5ad8af410019f17bc:action", "state_id": "bc17b399e85e2e7f48c7dc092990a7a3d6b7230937694fc022fec15ad3981d31", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.47265625, -1.9375, -4.53125, -1.55078125], "student_probs": [0.384610116481781, 0.2416248619556427, 0.018058858811855316, 0.35570621490478516], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b66e7c282558ad2b71042ef142686bbea072e7623a9766ecd73f7d43411aeafe:action", "state_id": "25c076c02d649dc620039fbebf71853c2e7134538c9ad6ae7be835ea7bce3b23", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.5546875, -1.796875, -3.71484375, -0.41796875], "student_probs": [0.6723656058311462, 0.06402283161878586, 0.009405277669429779, 0.25420626997947693], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dc5957388e0e53e940d87d668585d2bd4a63ce014326a1e6e6a2f328253a5aea:action", "state_id": "61ee81d4842a237d7d9b9985506914dcd3276be94b81378243dc3126a21c6476", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.08984375, -1.12109375, -3.03125, 0.609375], "student_probs": [0.3307681083679199, 0.09854172170162201, 0.014589817263185978, 0.5561003684997559], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7ddef82dbec3d1b513f8ec7ab94a58855d987f4b992ba3335d870545c785d7f1:action", "state_id": "6829225cb3bd8b996e99eba3d37b71b481692192543eb1d7d6ea5908e3d987af", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.439453125, -4.40234375, 2.7265625, 0.3203125], "student_probs": [0.037216782569885254, 0.0007074198802001774, 0.8825146555900574, 0.07956111431121826], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "956b5f9172a1026628d71a6944306c057256c217bd2d42df80c7760204316312:action", "state_id": "0c35b9d8e36b118c44c41085e4079b42ef160644c9d52a7a7e3719d1f13eb473", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4013671875, -3.1953125, 2.96875, 0.1875], "student_probs": [0.03130374476313591, 0.0019151431042701006, 0.9103734493255615, 0.05640765652060509], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f1847b9e8d6ad41b39e096e779334ae8f480e5ec2a233036640e38d9c485d08a:action", "state_id": "03b009823947120d3c42ce3414365c3f91e20c13777e8b1a75b76fee5d07ac4f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.142578125, -1.34765625, 3.4921875, 0.53125], "student_probs": [0.024298755452036858, 0.007281573489308357, 0.9207520484924316, 0.04766766354441643], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "088c09393afb7815c0b8a79109b3215ad3b39baf1c96db1d23ba5de417707e46:action", "state_id": "888efb586109cb67414568ed32bc511fb5c95499d07b7b399923d953eb74a7f3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.1328125, -1.27734375, 2.4375, 0.6796875], "student_probs": [0.07696454226970673, 0.01878744177520275, 0.7712652683258057, 0.13298280537128448], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4756815f0135a664db8870e04492b0bc63b65a729d9b1258f1409f69ea446a71:action", "state_id": "ef537555858ab70845dfcaf7c0bf457b76be0689ed626b0528324fc325366c12", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.361328125, -1.130859375, 2.103515625, 0.34375], "student_probs": [0.06557859480381012, 0.030377980321645737, 0.7713120579719543, 0.1327313929796219], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f5bd807606b0415b7a81c754bdab1e8bca91218ebc87bc9845814877312d477e:action", "state_id": "8768f7e833673b6ee8e9b5086a4d9752e205fd3feb301719a8c93173c2868fbd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.126953125, -0.4296875, -2.58984375, 0.6640625], "student_probs": [0.2481624335050583, 0.18334123492240906, 0.02114054746925831, 0.5473558306694031], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "53cd1142970158bcc0118a601f63143b6843f1f462b90dde4d23d492c754d2b4:action", "state_id": "56d7e9458a2c6f6d084f78e341551ccf322e2876ed1a43df01e60a82f9b44af2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.09765625, -6.04296875, 3.65625, -0.958984375], "student_probs": [0.008460772223770618, 6.021269291522913e-05, 0.981759786605835, 0.009719287045300007], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "15d52499d8ccea9c222b08df663456bf202f8d3b1ebfdff4a5d17e8fead7748b:action", "state_id": "c1c3fd91f23898e41ffca6e86bbc856eef2ae2760935060dcbe63edffe0a2469", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3203125, -5.8203125, 3.125, -1.20703125], "student_probs": [0.011447206139564514, 0.0001271669752895832, 0.9756053686141968, 0.012820262461900711], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "37a1b247fee95b934f7a4df2bd4c332d0fa677165e02ab8054b1f47e72c0e975:action", "state_id": "63ed3c70aa03cb9086e800dd3dec555ea99d2c4b21e447cd2ec16ce5f638583e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.0390625, -2.0703125, 3.271484375, 0.34765625], "student_probs": [0.03332953527569771, 0.004371883347630501, 0.9132327437400818, 0.04906581714749336], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4f6004aa6d9ffbabf852655d1a7bab763b541d45f7949503b930071098e12cea:action", "state_id": "00b0a13060b70142139783cac99d7e571f236d87fc984d78893d47dc93ef9d68", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.16015625, -0.48095703125, -2.60546875, 1.203125], "student_probs": [0.22587276995182037, 0.11896847188472748, 0.014215698465704918, 0.6409430503845215], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a119359ecc4a7a22b18d622b79f4d0265096d3dc8013aad1b2000a4426bcafcc:action", "state_id": "8b214d7fad3f33c7394d6d5fc57587ab34b17ffcb0625b32734acf21f0ec5469", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.05859375, -6.0390625, -8.1171875, -4.62109375], "student_probs": [0.789431095123291, 0.040078651160001755, 0.005016431678086519, 0.16547374427318573], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a65146b4c47266bbb923c8145e397c8f2ca7c0fbdaedecf2efa48aef4cb688c9:action", "state_id": "b71735763d31643a19b3b7b2543dfd2289a3a59f0367c8208e1159816ecce03a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.609375, -4.69140625, -6.7265625, -3.58984375], "student_probs": [0.6595861315727234, 0.08223502337932587, 0.010744833387434483, 0.24743399024009705], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dfdf35a248e14456cc226e662e153e5fecdbbf7596089c6b9440112a166c03ae:action", "state_id": "9ab380f0dedc2ae22703996cd8572fe0e0e146335352269a7d22be3b4e5ac508", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.88671875, -2.51953125, -4.1875, -0.873046875], "student_probs": [0.9278082847595215, 0.011320045217871666, 0.00213529821485281, 0.058736395090818405], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8f69bc436d004b6848171ed285475655c1f5ed97be8f8f0ebf1dbed774b06eb4:action", "state_id": "7580be7eca3b7a5eda0385fc0c5bacd7004a9ec746761e2632c2063917f2c95c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.330078125, -1.6640625, -3.53125, -0.00390625], "student_probs": [0.37177571654319763, 0.09793524444103241, 0.015136648900806904, 0.5151523351669312], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fa1a580bbf55260e252d1d5ac1a50e4088e1a6ad99bd276c346792e62497092c:action", "state_id": "074896edb1f57c3861d1b2bc4284a9afac1b083297a73afa1e0d14ce922ee7a7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.07421875, -1.26171875, -3.1015625, 0.484375], "student_probs": [0.3224101662635803, 0.09832954406738281, 0.015618884935975075, 0.5636414289474487], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5e108efcc8e08c312e9b81a0d153eac19529467d4335f46e1cc4af8e0597c5d2:action", "state_id": "5a0cd9af7e90f9a0107b9ec975f74b1b665cbb5db55cf65643fea49ac12ce9e6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4765625, -0.943359375, -2.3828125, 0.09375], "student_probs": [0.2821301221847534, 0.17689768970012665, 0.04193489998579025, 0.49903732538223267], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "252d3586738fc22291421ec65480751f37cef5e32cc7c858bd1d38427df7de89:action", "state_id": "0f774a5b20ee534631baac4815e9c5477f3cdcfcfeed68fec3b2167fb6c8a199", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4462890625, -1.3828125, 1.154296875, 0.11328125], "student_probs": [0.12348966300487518, 0.04840649291872978, 0.6120067834854126, 0.21609708666801453], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ab98dca54901b7704e0d4276f2b7c0528adc683a12f2fbeb5c7f95a4d71ce8c0:action", "state_id": "f5ebd38f8a437460f11d4b88dc6df8252d1ded1b7ce96b4a9c8e448a71d1865a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.68115234375, -2.140625, 1.15234375, -0.5779561996459961], "student_probs": [0.11632252484560013, 0.027028560638427734, 0.727681040763855, 0.12896782159805298], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6127a5be52ff6bc67c01fd3eb113e9119f73e14111ce062a4a1bc33fd605ba9d:action", "state_id": "5e5ae48a759a70c562d150086bd8ecb64aee091eb11e0db0d5f2f247ab8d8a3d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.49169921875, -1.95703125, 1.1015625, -0.2578125], "student_probs": [0.13487499952316284, 0.031156295910477638, 0.6635538935661316, 0.1704147309064865], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e5a03821e3b64a711517e1c9c524cc906e8d2541e2ef36ebc1ceb329e1374783:action", "state_id": "0412b538b1cdd6401c8b163fb8cc36e9c1168f63600fd48a9f10205f6a8e1dbf", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8271484375, -1.921875, -3.9296875, -0.59326171875], "student_probs": [0.3783482611179352, 0.12660709023475647, 0.017001066356897354, 0.4780435562133789], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8d726bba40a3b368e9c383dd6f1c1221a0bd8981b02c5b8f5bf94e1de8d87c15:action", "state_id": "19515a24253e7b2cbc8edc5139cfc13bff751a125bdb69e7f99d157bb3c6e8f3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.71484375, -2.10546875, -3.87109375, -0.150390625], "student_probs": [0.32786983251571655, 0.08161325752735138, 0.013962382450699806, 0.5765544772148132], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "47545422ded02ec6ec919c2017b7087006d725ac5fdee1c8ed59fc093f7b966d:action", "state_id": "d59f38a892e1e8f7e215549421432bd7e01534acefd779d9d3f86a48253fc2b5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8046875, -1.95703125, -3.703125, -0.873046875], "student_probs": [0.4338527023792267, 0.13705213367938995, 0.023909302428364754, 0.4051857888698578], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "94eddb7f98f6dfaa045ee978787d9eba22694577bb557cb111603a03714b64e2:action", "state_id": "c1ee1ea17513c2192068c5e19f705627eb70d5d593b42c117b7a775018efdb4e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.125, -1.48046875, -3.5390625, -0.421875], "student_probs": [0.5539616346359253, 0.1112329512834549, 0.0141970319673419, 0.3206084072589874], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5c7e7fc4412297b777b2981f8821ca1ac3dff6e71b14604684b0ed3010118124:action", "state_id": "dc814a849623896e5857e6dafc1a129bfa1b8b3476062fa915d9b93113ab9aef", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3515625, -1.5859375, -3.640625, -0.162109375], "student_probs": [0.39418280124664307, 0.11471373587846756, 0.014698600396513939, 0.4764047861099243], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7b9aeadb7c2057dda316df166aef8c962ea9320c57e2072d3469871ee247dc7f:action", "state_id": "bdf7bc50007f030ee9e5f5f6db91ab9fc5deb0d712b146b5e7a5e2267f830fba", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.650146484375, -1.9296875, -4.171875, -0.1171875], "student_probs": [0.33204057812690735, 0.09236204624176025, 0.009811239317059517, 0.5657861232757568], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d2874d7dc8d24721dd88ae1ce5410d0da3c3ec4ecc4270313d766e10db7da5eb:action", "state_id": "b9f6676f50e6b01c2f0b5a52952cb971dc11b8d01a309ca6d9280b1b8c336c76", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.99609375, -4.53515625, -6.2578125, 0.1796875], "student_probs": [0.2339235544204712, 0.006793266162276268, 0.0012132171541452408, 0.7580699324607849], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a976280a658ead7a836aa453e5c818e1472cb0bc0fa95ecb5457b5887a87199f:action", "state_id": "64aee92a89a5c301ddc748e9769d46ed015517c62cd680d69afb385e5a50f37e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.9453125, -6.0078125, -8.15625, -4.390625], "student_probs": [0.7764580845832825, 0.0363154299557209, 0.004236786626279354, 0.18298974633216858], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5dc53e802244160c12d90d038d3a187be48cc05d1c36368a6d2e9c16140b1e54:action", "state_id": "47924f380260b539da0aeaaed833728de4b069b3f8290e3c0f101e2d9fe35e8d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.51171875, -4.42578125, -6.5625, -3.3984375], "student_probs": [0.6341579556465149, 0.09352563321590424, 0.011039909906685352, 0.261276513338089], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d61e0f5ea1b0e6284fffac4e0d30bb61be6999f2be01291454dca1aa50407d28:action", "state_id": "dc90bee649f44c6ae63b79d32dcf03de8bf6c6104e32b4c18ddb6301e542b476", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.23046875, -2.65625, -4.6484375, -0.5361328125], "student_probs": [0.6544702053070068, 0.036492519080638885, 0.0049774604849517345, 0.30405983328819275], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a8e0f2a9c473bebbf14d147626a347f43aeb8e9e0f744765d8e26b2bfd46b1f6:action", "state_id": "883011f814b4e1fb57931d90a35779f23545917b6ef5c15a02d9cecc6b9cc600", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.02734375, -1.59375, -3.3671875, 0.6328125], "student_probs": [0.32644006609916687, 0.06453144550323486, 0.010954114608466625, 0.5980743765830994], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "47fb1f3cbbf284109c1d0268b870d7c6f1c6467cf1485be09f86bc82d3aa65b0:action", "state_id": "2c36b10645dcca15bcf74dc60cea268885160a10f107f588bc07840f4bc0fa33", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -4.51953125, -6.1640625, 0.171875], "student_probs": [0.23002828657627106, 0.006987072993069887, 0.001349225058220327, 0.7616353631019592], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3207f0b5dcb7847dececfdf4ff8ccf93770b8895afedbe4242d0c010baea030b:action", "state_id": "73076e766c0d6879bbbdf8deb1ec32e462b19c740641ed39dee4bb028bffae99", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0703125, -4.58984375, -6.21875, -0.0078125], "student_probs": [0.2545165419578552, 0.007537078112363815, 0.0014783525839447975, 0.7364680171012878], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5041b69fbcb99981224fc9261aedca64a8d7439ed37a7f82b8d2cb7b9b100c73:action", "state_id": "2de1dbb20767ddaac952033c75e895fcb9d30a49a50298e9357e8bd46b4c775c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.873046875, -3.3828125, -5.5703125, 0.09375], "student_probs": [0.2688232958316803, 0.02185191586613655, 0.002451716922223568, 0.7068730592727661], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "59daf5bd6c17f6993fbce0535d410bf5deac820d0aae3710e09148d5386443fb:action", "state_id": "c1c25e3ffe10380556699e657a5ea348e740cd52509fd40cedb1bdb9cb9a2289", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.625, -6.203125, -8.1015625, -4.875], "student_probs": [0.7279115319252014, 0.05526028946042061, 0.008278129622340202, 0.20855015516281128], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "516a7a2ca70cbecb20781089fcff5959449330bb8dfafd6ada7c5a3248ccd814:action", "state_id": "f52f55d2c5874319946ae10408b8cb6f266f39ecd320ff8f95a44ed47c3efd18", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.48828125, -4.2890625, -6.5, -3.1875], "student_probs": [0.5951511263847351, 0.0983009934425354, 0.010773577727377415, 0.29577428102493286], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c3cb5a03d51f43c0e962d95ea45cce54fa50adb2d3494b31cee847412b8a633a:action", "state_id": "58729f20798c5012b30283bbb28eb1fd573ab36a89a895b09e23c4629bf184fd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55078125, -2.5234375, -4.71875, -1.91796875], "student_probs": [0.4732951819896698, 0.1789422333240509, 0.01992052234709263, 0.32784199714660645], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a8e9758cf7c1f3f763d1c87a5067449514d18afd77f2c6a8135785cbadee74d0:action", "state_id": "03bac26bcef3d5774b940ba1b5e6cf0ce1bd7aa3b287f19fbc0bf0c60fb3a055", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3203125, -1.81640625, -3.5859375, -0.287109375], "student_probs": [0.7992143630981445, 0.03470592573285103, 0.005914336070418358, 0.1601654440164566], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c7f82e35f4579919bee8eb960ff621e76078922d1c0edd2680811409199420ea:action", "state_id": "4b3279d698d3d51833d1394157eb48b20066e5b4d5153e35063382251b5bad44", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.49658203125, -1.8125, -3.875, -0.4794921875], "student_probs": [0.4311150908470154, 0.11563713103532791, 0.01470161136239767, 0.4385460913181305], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "21c1b36c7a27285b8d8d3aeba7488dd0a36011682aa75a53bcc50830fd2f4236:action", "state_id": "4de31e86d96324dec68e0c276ea3d76a539c3f72264f305285a40b7eaa437de3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3984375, -0.7548828125, -2.33984375, 0.296875], "student_probs": [0.2598753869533539, 0.1819545477628708, 0.037292640656232834, 0.5208774209022522], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7f5ea75e800be086ccdc09c9ed1077e36c6737cbd9b60a6370ff8ceb0f04257d:action", "state_id": "ecfb116926c59fa970b657d9d16813bce026afdda504224484391d6d96c61708", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.46484375, -1.26953125, 1.560546875, 0.27734375], "student_probs": [0.08987289667129517, 0.040193650871515274, 0.6811531186103821, 0.1887803077697754], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "57381869f9b7f4467a4fbcc410b096c507b0ab02d85e041451c7f8dfde44208f:action", "state_id": "110f710da3dde06daa9626208b1d1d5b1e4f996f61e2ceb516873c9744f6dc99", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.708984375, -2.12109375, -0.4775390625, -0.328125], "student_probs": [0.25204211473464966, 0.061404723674058914, 0.3176790177822113, 0.3688741624355316], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "96b93b5002b1bfe6b2075ce90b75229bcea485b4d484ebcb03c70486880c04f8:action", "state_id": "27d1d57b866dcfaa81a1ef29ace101708046b49687d9bf7d193872b98e53a1e2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.70166015625, -1.421875, -2.90234375, -0.125], "student_probs": [0.296080619096756, 0.14408694207668304, 0.032784249633550644, 0.5270481705665588], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "17684c5da20ee44d326de543558e813fe8bfe3731ba3469ce0042c63e6ed74eb:action", "state_id": "6e5afd19602f604be9526d507c25a6f1f06b86c7c5f4bc3dff275df2e023f5e6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78125, -5.02734375, -7.1015625, -0.837890625], "student_probs": [0.2768232822418213, 0.010775613598525524, 0.001354004954919219, 0.7110471129417419], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "030ea486ebc4646395e8a222f4d36e9790e4f1e8eccb6f0ce4fdcd69c4bc3e55:action", "state_id": "03c38092c9114bc0e6715e86d9296f453518d957342ee6f0fedd2a51046eb867", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.578125, -3.140625, -4.9609375, -1.19140625], "student_probs": [0.36822840571403503, 0.07718487083911896, 0.012502028606832027, 0.5420846343040466], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ad7f97b8730e6f793d16f21a41cbdbf6a3164191db67674975d0da190256c76b:action", "state_id": "1417a748be34e91be0267ef488860fab880b557a8ea049622aa4b31eb9cb547e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.23828125, -1.75, -3.8125, -0.6346435546875], "student_probs": [0.28535255789756775, 0.17105869948863983, 0.02174767293035984, 0.5218411087989807], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d28e441efa7cefdd21cabb1e88ef4dc884061bdd4983482c416b32380495f3e6:action", "state_id": "23f178ab95044946beaba92071a77a57b79ddc1597d829b0f23f21df506cf65a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8642578125, -1.48046875, -2.7265625, -0.00390625], "student_probs": [0.2463483214378357, 0.13302479684352875, 0.03826141357421875, 0.5823653936386108], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "46fce84a931297f5d1354d625db079c54256a977794479e131d31027e789852b:action", "state_id": "8098508e183f8da4b86431604b0f6d84e05e3a1799d9c052001a9cc7ff564bdb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.25, -0.8076171875, -2.42578125, 0.734375], "student_probs": [0.2292405366897583, 0.13125666975975037, 0.02602325566112995, 0.613479495048523], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e3e5645aab931160462d03d0b72c5f25f3c9f19c463d89988b36b413defe9a8a:action", "state_id": "97dcd8db304d6ef7c59d32d66872b4cb475c0e471f1758889b5a20ced00d0acd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.2421875, -6.0, -8.0625, -4.66796875], "student_probs": [0.7623024582862854, 0.04835312440991402, 0.006147409789264202, 0.1831970065832138], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fe606504bd381007f4a80d7ccfc0a6bc6df1055df6c9617a5b39062e7a8fa40e:action", "state_id": "ae29b018d36c62563dcc0d11056e427e5c8f26c24bb209f5df913e2a4ebc61eb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.6328125, -4.4921875, -6.4921875, -3.47265625], "student_probs": [0.6216473579406738, 0.0968339815735817, 0.013105054385960102, 0.26841363310813904], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2bd05a8fda65aca58002819ed6a15427218f5a00aba76b0e5e332b85a10d66a2:action", "state_id": "dffd285b269687703f188918582b2c8017ec31c3ef3f520a698863a5afb5e60c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3515625, -2.55859375, -4.54296875, -0.6943359375], "student_probs": [0.7075485587120056, 0.03853819891810417, 0.005297712050378323, 0.2486155778169632], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3f7f684a103b192006f71cf81992ab97bcf22b0805e14b653fbba4315cc13cf7:action", "state_id": "a5f47e600065d0dc09485f083633ceaa643a6eb28927fdecb8d95dd51e58d1f7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.197265625, -1.73046875, -3.35546875, 0.39453125], "student_probs": [0.32620275020599365, 0.0704086422920227, 0.013864283449947834, 0.5895243287086487], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3f991233ce5416f0e39ab04489192686023c84d57c179e582452313d8cc2336d:action", "state_id": "2d11acd0b6af68da0618e2732f39a145863de6d64fd47e3ff943e99d03e7129a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.09765625, -6.04296875, 3.65625, -0.958984375], "student_probs": [0.008460772223770618, 6.021269291522913e-05, 0.981759786605835, 0.009719287045300007], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f8bee3d6f3a3b090732cc42d497c03b72b74a47c60943e337b66d3130d1644a6:action", "state_id": "99bdc9e5e18d2209eaed8ad311b97823f342b07fdac55c66a163a64cfe2b6b38", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3203125, -5.8203125, 3.125, -1.20703125], "student_probs": [0.011447206139564514, 0.0001271669752895832, 0.9756053686141968, 0.012820262461900711], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5bd09491faf461e39903de076a1b428dfd11ab8343551273e6dabe537aa01655:action", "state_id": "48ba7c60421fd9b094a83629469439690a5927740a5e1ae69fb022baa21553db", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.0390625, -2.0703125, 3.271484375, 0.34765625], "student_probs": [0.03332953527569771, 0.004371883347630501, 0.9132327437400818, 0.04906581714749336], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7ca48662983d16232643ad1ca2d5655fac2915ea489061e2f92317d499cb3021:action", "state_id": "b968c139b5df18608208035618ff576b7d455fa4fa2afe0c681c98ccc2e2ddd5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.16015625, -0.48095703125, -2.60546875, 1.203125], "student_probs": [0.22587276995182037, 0.11896847188472748, 0.014215698465704918, 0.6409430503845215], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9b831d4e53998ec9141eb550bb4b5ccb5ffb7cfa8e6d8ab51db78e7cd6751047:action", "state_id": "11f0e601fcd69801bf55a03786cf19c82e36177802b76d8591994ee8d911319e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.98046875, -5.8984375, -8.0546875, -4.34375], "student_probs": [0.7598096132278442, 0.04106266051530838, 0.00475334795191884, 0.1943744719028473], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c7dbcda12833fd764a2d0bf2e12f65b945995b23bb513a84ee2bf18bf48a2b30:action", "state_id": "e68f263920aa6efb20e4365c9fa21df444efce34409e21b840a8c312c09723a1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.48828125, -4.45703125, -6.5625, -3.4140625], "student_probs": [0.643973708152771, 0.08991887420415878, 0.010951092466711998, 0.25515639781951904], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7a1deddf2c98686b50d226871a66580b87168cf4ac80974928dd56faaa8e6d51:action", "state_id": "02b1c62691ddb1e99f91ca4b5de580d19f9ac9d42792010ec5030ccbeaeb8fa3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.24609375, -2.53125, -4.5390625, -0.6124267578125], "student_probs": [0.6691895127296448, 0.041625939309597015, 0.005589618813246489, 0.28359490633010864], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ee03b56e55638ee599fb6279c02f92b61ab6cf701c5ae27c841d907d29720f1d:action", "state_id": "901e292767b9c10150d07c1c9869ca6c564e5b50f75bf60fe91e5d591132477d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.41796875, -1.84765625, -3.6875, 0.109375], "student_probs": [0.5391630530357361, 0.055946338921785355, 0.008886642754077911, 0.39600399136543274], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "17c98a6a45f2cc360b917f42cab55bfaba28eaeeebe43378791af3edb02c0da6:action", "state_id": "94c8bba61869f6a8bf617e2da72b25692887f3691ec13e52305c1e830814c088", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.13671875, -1.33203125, -3.28515625, 0.23046875], "student_probs": [0.3585260808467865, 0.10849335044622421, 0.015387630090117455, 0.5175928473472595], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "44889e5f4a0c0df866fac98d0db5ce00050a5918e77fa99cc4c7c95b4bb61b27:action", "state_id": "de899915571af584ad98f7ce3badffc2d6eeb9c07bb6437477ebd8ff8940cb4d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.310546875, -4.75, 2.65625, -0.14453125], "student_probs": [0.046249233186244965, 0.0005458515370264649, 0.8986034393310547, 0.054601456969976425], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d92b5d70bc881df9e31927e8b698439e5032859cda5e3974f5f4a5a93bd5554b:action", "state_id": "78b7679c6ba49acb3defcaf39e643c30043ab305d0d745bed05e831b2f88a542", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6380615234375, -2.8671875, -4.21875, -0.154296875], "student_probs": [0.362627774477005, 0.039026889950037, 0.010101544670760632, 0.5882437825202942], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bc90054ed4642c0b30a33792ab67790bdad39fad3899dc9cc6d795e6046c0966:action", "state_id": "4caa55d546ed5820dcaf11a09c32ce752d148e0e815344e2e1a693573fa250f6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.21484375, -6.1015625, -8.0625, -3.9453125], "student_probs": [0.8329164385795593, 0.017085235565900803, 0.0024043438024818897, 0.1475939005613327], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1732730641d3cb472256bed93b7defcfcc1e8e24d1babd63b0b24b0c00e17aba:action", "state_id": "01585a050b2688b8075c3fe70570911d3dd2683fff53b3251c92b645302c6ffc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.46484375, -4.33984375, -6.703125, -2.60546875], "student_probs": [0.9475361108779907, 0.007761515211313963, 0.0007304432801902294, 0.04397197440266609], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fab783d2f99ec02b781bd2ed53766ab8f14ceb2c683284d96d3d77df531f3ab4:action", "state_id": "8102c41c48a4f80a1a49d53b4057f912a24a4d355ab0da68061b837eee517b5d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.462890625, -2.3203125, -4.12890625, 0.0546875], "student_probs": [0.3496978282928467, 0.054578911513090134, 0.008944634348154068, 0.5867785811424255], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3f7e76aacb1ae1c5631bce87a0e771d1dc7b26ed353e20d5a88d4e182efc9cfa:action", "state_id": "f53a4d831a800df3c45aa49b805d2b4135d9409b765ebabb040736ffb24998a9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.21484375, -6.1015625, -8.0625, -3.9453125], "student_probs": [0.8329164385795593, 0.017085235565900803, 0.0024043438024818897, 0.1475939005613327], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2d3ddf62148b61fa83b9c8881cfd36aae66293bd0546a86b1be45bc1e4bffd6a:action", "state_id": "9c0ec10a6ecdbda86600e70346292fc65ed18bb2f61ea0f7e1d5a0e8b0b19783", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.19921875, -4.35546875, -6.6328125, -2.71875], "student_probs": [0.9384034276008606, 0.009869927540421486, 0.0010122227249667048, 0.05071446672081947], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b72b169c338bb21936a4b70a9c73837d36b161c0437346d7cd8c9eb6db738251:action", "state_id": "9fe0015021ddceb40397ae7706cce47b517993d26ae8715c663cfbdb946c66d8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4716796875, -2.328125, -4.12890625, 0.05078125], "student_probs": [0.34865036606788635, 0.05446859076619148, 0.008996566757559776, 0.5878844857215881], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "439f5ebede8f1cb3e82a7cb3c877954400d0100fd4c564930e03255b1d4c92fb:action", "state_id": "f21281db66c1c27343e3d0e65203066b591f2a4744fbf5a7aa2af47d096c3fd1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01953125, -6.15625, 3.71484375, -0.958984375], "student_probs": [0.00863106083124876, 5.072424391983077e-05, 0.9821484088897705, 0.009169789962470531], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4323e7e03dc6d2823a1f0325a981d4f74eeac18b60078de3d588f9dbdc3264d3:action", "state_id": "824c334cf2a8f7872242852ed56c042fdc2f0b74fb2d558581b2b99f12ab1f1f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3828125, -5.9453125, 3.10546875, -1.046875], "student_probs": [0.010943543165922165, 0.0001142061228165403, 0.9736295342445374, 0.015312770381569862], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a23c8f1891e5d8ce80b71ec8c2a319dbea59beaef4b4fdc9f5f0c69983e493ff:action", "state_id": "fcbff9973b2f915b492d9b029f5c1eba591681fec6c39b97f2a067cf496c6108", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.26171875, -2.109375, 3.712890625, 0.38671875], "student_probs": [0.017762156203389168, 0.0027994243428111076, 0.945467472076416, 0.03397101163864136], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7fd95b10faff561054760d47abb760abb7aa1729f2cd8b17a06cdc2c439eebff:action", "state_id": "43941397e7606861610d4ee74cc29c2645bfc29a64630941f61752ef30177c9d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.71484375, -2.2109375, 2.986328125, 0.5859375], "student_probs": [0.08601070940494537, 0.004612134303897619, 0.833768904209137, 0.0756082683801651], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1e2ee11df2a31c0aef9bec8e6a839dec1139e1bf77655ab2996eb51fca9d866e:action", "state_id": "91dc3a91993296fb02c173cd38eabccd1d398aeb5031b7fec134ef6d4ef38283", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.1171875, -0.66650390625, -2.578125, 0.59375], "student_probs": [0.31900298595428467, 0.1456940919160843, 0.021539490669965744, 0.5137634873390198], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "304d6ad45f0509357701cde9a1ebeb9a6d997e60fde58ea682d6f251c1d28901:action", "state_id": "86bc6cfd5c5afbd829e1991b71feb973f53898b9486e490a1002133926ff43fc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.251953125, -0.974609375, -2.40625, 0.3046875], "student_probs": [0.29884225130081177, 0.14507627487182617, 0.03466113284230232, 0.5214203000068665], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "868a8af868e23213e7a23bb1ed894a3c798f9fcc2c44a7df3afc6ce9916b2f1b:action", "state_id": "49aa98d1be05babd9d835c9f037f779de20ff2f51b983591c7af8fdab2bcfec5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4970703125, -0.712890625, -3.51171875, 0.16796875], "student_probs": [0.2631918489933014, 0.2121010720729828, 0.012913002632558346, 0.5117940306663513], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bbe2dfce7b0cb75b2ccf12fd69114ad280a16e0fe3e5162942d470846231eb91:action", "state_id": "d745a9f2872be33df5dca5c2fa1c0d50f140bfd2eb4edf917857a693d7dfe52a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.50927734375, -1.03515625, -2.9453125, 0.0390625], "student_probs": [0.2933479845523834, 0.17337912321090698, 0.025670034810900688, 0.507602870464325], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "673e91c2b761dd01c304eb31e38b351b19a406a8ccc982baf24ebc4b939b6767:action", "state_id": "fd2498a31543f3cbaf661ab8ccb105bf707cbe4e971d9b68f85f0cdeebbc6832", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.455078125, -1.048828125, -2.1875, 0.25], "student_probs": [0.26644548773765564, 0.14714518189430237, 0.047122370451688766, 0.5392869710922241], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6ccaccfa277d44c8ee58875ec910d4ab49adc0ee47fb56e59266688e51e1d1a7:action", "state_id": "90d461c63fba7ca63e26a50fe19e5780676429be09bc1c6d1d43451ba8ec03ba", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.189453125, -1.52734375, -1.103515625, -0.7313385009765625], "student_probs": [0.2280968874692917, 0.16269542276859283, 0.2485659122467041, 0.3606417775154114], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "38c2d7975bc3738538b1e4ae1e5afa476cd8dff22a32e480f52ca67dd18d8b98:action", "state_id": "09e1551b5ed0390d37b51ebd0589768e6b48e765e017ec1f7aa21c6ebd220bad", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.876953125, -1.060546875, -3.5390625, -0.111328125], "student_probs": [0.2467665821313858, 0.20537738502025604, 0.01722451113164425, 0.5306315422058105], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6dcee63fd5c4cefdbe192efc51e60f57cf37d670c552956792541dbef41df5f8:action", "state_id": "23d057f05f835a740b0dba74392cccf6c41f208cd1f9481aba8b2dcacf11958e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.609375, -6.1328125, -8.0625, -4.875], "student_probs": [0.7278611063957214, 0.0583624504506588, 0.008473852649331093, 0.2053026556968689], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1078f5045aba83edf829ad4914b1fbefd173a543f9c3ca44b29db408b6dafac9:action", "state_id": "2bf7ec8b97cc8db1de78db94de46d9dfe88717b5575fb40856a8c6922dbf6965", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.5078125, -4.23828125, -6.46875, -3.1875], "student_probs": [0.5871915817260742, 0.1040511280298233, 0.011183210648596287, 0.2975741922855377], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7a707f27d7db1e151bfb68a7926bd4c6165569fba010d506b4766304dc5eb12a:action", "state_id": "a5ce92471b1a1953293e249f393ec749b348701fb171a634ce9af64b76b1d514", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.1640625, -2.33984375, -4.734375, -1.6171875], "student_probs": [0.7365328669548035, 0.08361078798770905, 0.0076265945099294186, 0.17222966253757477], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6374eaa39bd32d0ab233b84bf6ec217e01e38eaa13c388c348ddace9ea2d4b09:action", "state_id": "768dedbbc62255430ee8d09fe0ea84186b48c09925766b7a5428fd3a78f2fde8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.171875, -1.63671875, -3.37890625, 0.140625], "student_probs": [0.37901344895362854, 0.08759535104036331, 0.01534117478877306, 0.518049955368042], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "84fa31ef1d1853c9710c82151e3a1f8af49e63ef94766b6387bc63a2d9340fbc:action", "state_id": "070ccce9f0ec42a9eba8b1d47753b3e6178d2ef7b77a33e8a85467893f135b50", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.41015625, -5.8515625, -7.984375, -4.56640625], "student_probs": [0.7082069516181946, 0.06164117902517319, 0.007304697763174772, 0.22284719347953796], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e041940e37f830c0b2d6f19ec21c20071165a9fe63c8682e4efdf91fdb5f837b:action", "state_id": "14bc9d85cef92b22705638639c1b28a817835c276d92e47a64f3366416e014c8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.91015625, -4.21875, -6.3828125, -3.21875], "student_probs": [0.4912283420562744, 0.13272976875305176, 0.01524501945823431, 0.3607969284057617], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7db635fef9450dedf88e6f0ddfd5ad302896ed7b208f122c4f17a39397a859cc:action", "state_id": "3b72faf71c1904d232c5028181c315e766d65127da06cc3c1de24091a67df414", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.71484375, -2.6640625, -4.7734375, -1.96484375], "student_probs": [0.4519166052341461, 0.1749112904071808, 0.021219147369265556, 0.35195299983024597], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7a7016305ba3ae381a7b5f8a4c1ed723da74719795789bc7d3c22627b5821d38:action", "state_id": "5ee19a534a0e7e3df81a84b063c80b092f0273bfc83058bba5cced1eb9ddf6f0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.740234375, -1.83203125, -3.76171875, -0.4306640625], "student_probs": [0.36400946974754333, 0.12216627597808838, 0.01773775741457939, 0.4960864782333374], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7777a91b51791d54d6c1b9a56c0b618a189d68603fda766f9bfc6e678a34ae3c:action", "state_id": "2120031832b71406ae45eb4af7600bc61cc2f2328543e2d541294252190bb049", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.9921875, -5.96875, -8.0859375, -4.41015625], "student_probs": [0.769640326499939, 0.039226822555065155, 0.004721720702946186, 0.18641112744808197], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9ac5064d5598936bf321763fe847bc916c3c0d88bdc00f3c6d765b3a2fc5f117:action", "state_id": "3d70e559c4b3bb49b7c80041d249096965183d54b3a8f630411f726b4e8d7238", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.5859375, -4.53125, -6.6640625, -3.49609375], "student_probs": [0.6400642395019531, 0.0914924144744873, 0.010842174291610718, 0.2576011121273041], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3449af747c033a9a13a54ebdeb7d61e024269f9f0c8d1b8f66de39ff133ae983:action", "state_id": "62d75d2de559abde8893d2f23e43b905878dc6e594b608e75d237ab542d001da", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.40234375, -2.5546875, -4.55859375, -0.51171875], "student_probs": [0.684991717338562, 0.03560106083750725, 0.004799296148121357, 0.27460789680480957], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4bac0b9198c5617193db02ed9515230986afb41ac2ad61c9d401c8a46f691885:action", "state_id": "1251f2303ac6ce070308922a3ec1c4bc6d393a14ddd73ab75f77260a30548564", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.107421875, -1.7265625, -3.34375, 0.49609375], "student_probs": [0.32616716623306274, 0.0646035447716713, 0.012820967473089695, 0.5964083671569824], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e1999ac6ed1637aa572ffbcb9c6ec124932725bc915087724e6c42cda4d6d447:action", "state_id": "b4bb56d91b300c079b09c72614764e22cba40fe27da2dd7e076ad86895e862cc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.640625, -6.09375, -8.03125, -4.78125], "student_probs": [0.7052004933357239, 0.06066441163420677, 0.008739536628127098, 0.22539561986923218], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d2a101a0a26ecbfa619d08af76281d2408343368429a05a304cf3ef2a28240bf:action", "state_id": "cb25791bc4973b0a5b9053edbfda17657f102f194f094f8f7315b20a43aea192", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.66015625, -4.41015625, -6.6328125, -3.26953125], "student_probs": [0.5759413242340088, 0.1000835970044136, 0.010841154493391514, 0.3131338655948639], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "34187c6395d2bb50e07599d542db87248fb03dce79fbc3fc54e74bfacb6583ac:action", "state_id": "8772bd3cf87aeee0af24e07d2c74851b586e65682e3d281d9394d1aec65ecb4f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.01171875, -2.34765625, -4.7734375, -1.64453125], "student_probs": [0.7688463926315308, 0.07436264306306839, 0.006574328523129225, 0.15021666884422302], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fad3f2235fbbc491111f33e38e862df2d113959be531ee9ea5d7d5e9115f781c:action", "state_id": "c2b0aaba6b331b44460c7de63f213edd73820bc83f8e2d6cd1d0ec2e4cbaebe8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.263671875, -1.640625, -3.39453125, 0.015625], "student_probs": [0.3819379508495331, 0.09638060629367828, 0.016683142632246017, 0.5049982666969299], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5e266bf0fb4c86a109d65b7958264f2f08f0e6e308f228b3a9217983f95a60c1:action", "state_id": "9517c1bd1c9b986b57cc3a5d6567c5d1348997e3541241f858a11e41f0efec8f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.39453125, -6.3515625, 3.42578125, -1.18359375], "student_probs": [0.007921060547232628, 5.57149869564455e-05, 0.9822419881820679, 0.009781205095350742], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ad6bfe9a162d1622451f21004290d53b981856205c6a64cf21e8b5b1ea3524c1:action", "state_id": "3f1740bf148d98ac34eece57d0a97c43cc71c79534590eb6c7bde020c9c2ae0c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -5.859375, 2.83203125, -1.046875], "student_probs": [0.020466569811105728, 0.00016122455417644233, 0.9595353007316589, 0.019836880266666412], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2fe71f85972798531d7da94a93b17ae78cb772e5b1b276555dc781fb941919ed:action", "state_id": "adafd00386e6fa0b1ff1265322c0e4c07554005359342165047842bb7e3238ba", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3125, -1.005859375, -3.16015625, 0.39453125], "student_probs": [0.2788750231266022, 0.1394079178571701, 0.016169188544154167, 0.5655478835105896], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "268512f18b47145da22739b1389092d501d2065868e863278604098353697f73:action", "state_id": "de02482cd27452103fedf80d29d4d087967bc8009778e0910baccb67a23c58fe", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.005859375, -4.53125, -6.25, 0.23046875], "student_probs": [0.2233305722475052, 0.006574920378625393, 0.0011788182891905308, 0.768915593624115], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "92d76e9f9c37615684e1bb32d5b55665844a6e5f7d739ab8dbbcdda0fa459bb2:action", "state_id": "87467b4752587ce60749c91a2a1e2941864fa057b8c3306acab94943c46f188e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3125, -6.21875, -7.9921875, -4.11328125], "student_probs": [0.8412549495697021, 0.016922511160373688, 0.0028725704178214073, 0.138949915766716], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f4c0f7f74943cb67a0ae7b5b3449d90fc51f167e4ba57296134760045d929ee0:action", "state_id": "bb760237d1c6873509a60b4b80191057e2c3759aa895e131b41ff41dccdb3f66", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.67578125, -4.48046875, -6.671875, -2.7265625], "student_probs": [0.9618135690689087, 0.005543192848563194, 0.0006195043097250164, 0.03202372044324875], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "92c4da7b23b308e61f2ac356089bf7e7ecb9310789b47e20bf0bcb6e632b12d2:action", "state_id": "73c174b0b6b453d6fa14c50b31ec43215e7df39097b49f21209af04e088a7613", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.04296875, -2.46484375, -4.41015625, -0.54248046875], "student_probs": [0.8070365786552429, 0.024180740118026733, 0.0034564565867185593, 0.1653260737657547], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "359b435c823190308084e1b5660d83f7f2850122c35825bf7619050388678ccc:action", "state_id": "49e6bfcb0e927bcfd4cdcbca6fbc2559d08584acc6a59410a16af44a24f725c1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.1484375, -5.6328125, -7.8671875, -4.33984375], "student_probs": [0.7162822484970093, 0.0597219318151474, 0.006393771152943373, 0.21760207414627075], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9b9c0c89d0678f612f3a6f3acefb82d13dc92dde8a4633887fb626f7e199f9cc:action", "state_id": "d4afdbda3f2498063e7529044c3db3e2d79a15620e0d484423028d829f6303e9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.7734375, -4.04296875, -6.328125, -3.1171875], "student_probs": [0.49537816643714905, 0.13918308913707733, 0.014163014478981495, 0.3512757420539856], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e18578e598df2866eac5f15e5729a1132780ef542535829fafba1f43e6171719:action", "state_id": "24658b733a55bf1700cd12c4cac9e5443fe6b4aed16dcc42b2727a37eb86e8c5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58984375, -2.55078125, -4.5, -1.73828125], "student_probs": [0.4349612891674042, 0.16638751327991486, 0.02369113080203533, 0.3749600648880005], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4f7b3e07d41b06f7e7035649b3e4645ada13b8e1da14923fb52d4c4671d7a4bb:action", "state_id": "1880742fb0496369af255147fa81d384fc95788b63c5323053c34066c0d909b5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.645751953125, -1.66015625, -3.421875, -1.087890625], "student_probs": [0.48366230726242065, 0.17538484930992126, 0.030122244730591774, 0.31083065271377563], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f93ca2ac221509d5f98d533342bef6cd40fe543499face19ed4c21abb487ef95:action", "state_id": "8b6fe27df334e1b1c989d5d48d4b6a976bd8c7aa95846e720234b229d0cea21b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.69580078125, -0.8515625, -3.375, -0.37109375], "student_probs": [0.30229687690734863, 0.25869452953338623, 0.020743029192090034, 0.41826558113098145], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b9248570456d26fb1d00cb734604812a33e2bc4a4f684793132162065a86b549:action", "state_id": "90d42876dfe0f9137a3c38f14b7236ebfaa457c35433d307f20e1c44b32b3ce0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.310546875, -0.9375, -3.26953125, 0.3046875], "student_probs": [0.2910209596157074, 0.15546834468841553, 0.01509571447968483, 0.5384150147438049], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9bcd350fd541b9852629f9b5f2c8614e21d174e02abba2138de81b5acbb1548f:action", "state_id": "2b86843b026d79d4f8742688cb6affe517e0d654131a03b03190eae93f4a92fd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8310546875, -1.5, -3.0390625, -0.138671875], "student_probs": [0.2761942744255066, 0.14148011803627014, 0.030359113588929176, 0.5519664883613586], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fece032fb4d0a550f468f030aa1076a8fc14015f1493cf801f81b1c7546c0d55:action", "state_id": "0e4026a778320ba023373e090d305e809a791a885da6ade9df3044151edb0add", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.744140625, -1.5078125, -3.2421875, -0.14453125], "student_probs": [0.29677069187164307, 0.13828100264072418, 0.024408048018813133, 0.5405402779579163], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "42e268756bf8f59fdf88ea30735403855117c18f39c667f9248762a603b2a609:action", "state_id": "b4a4298ec2fc3f11d877f10d54942eef3fc6f94bc86c24e8e0f14a7649b81d7f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0546875, -4.5859375, -6.421875, 0.13671875], "student_probs": [0.23117955029010773, 0.0067662340588867664, 0.0010789702646434307, 0.7609753012657166], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "eddd23bf0989f3676dd6623740bad038cdf578dae5e653f104c8b74060066505:action", "state_id": "d691bd0f47fbc90346c9be101ad470b9a674264b77e1544b35d787ccec2c7138", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.96875, -4.5, -6.125, 0.1875], "student_probs": [0.23735912144184113, 0.006947099696844816, 0.0013679651310667396, 0.7543257474899292], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4a8ef0f720f638c71ba94120da8947dce84a0043e654a76999ac5f78d2af5e2d:action", "state_id": "239bb8520c654f8758e4079c68c2befd16643ad44bb537467c207be4020119ff", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.064453125, -4.58203125, -6.375, 0.21875], "student_probs": [0.21538777649402618, 0.006390816066414118, 0.0010638486128300428, 0.7771576046943665], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "23ee766dbf44242162856f5e74afa93810f788dcd8eb5ef7fe8c02de1f84147b:action", "state_id": "44223f1820130d87d787b2b28884ea6d24f6c81dd6bc37e8a91c49cb5250555a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -4.55859375, -6.4609375, 0.25], "student_probs": [0.21410632133483887, 0.006352792959660292, 0.0009479541331529617, 0.7785928845405579], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ef0dacf8ef626ae7055d86937d3b8b6a0bd8817df0c66da0374fd27026a725c5:action", "state_id": "6f5f2715709fff79b63a2af2eefa76f58b8f60ad3b196cbca5b0aa2d31711b92", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.1328125, -5.6015625, -7.875, -4.3046875], "student_probs": [0.7126588821411133, 0.06035555154085159, 0.006214065942913294, 0.2207714170217514], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "31906029697a7d7ddd1415e91c6da255c8c96f4dc6490cb0b6c9972fb7f29f7a:action", "state_id": "ee5856b487344a6b14b8db621505d6b5c300d5c61df10e3959336c11ab78928d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.828125, -4.015625, -6.2578125, -3.171875], "student_probs": [0.48864251375198364, 0.14902755618095398, 0.01583058200776577, 0.3464994430541992], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "239cdea5f0e5648edb805dad4d345aba47a35853c0f492381ac76362db8b747a:action", "state_id": "e33812177229ddf9a73125a49ffcd8566c23730f12ce36aea8d94bb8b575ef57", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.90625, -2.796875, -4.671875, -2.01953125], "student_probs": [0.42261219024658203, 0.17343969643115997, 0.02659783884882927, 0.37735021114349365], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0cf7edd256a11b9d5d65028d894c26cfedc991e6dd64c393bb329342d8a918a3:action", "state_id": "4a4be6f27e3bca202c1d83d2f91b032b2dc0768231a183cb21c45c67f2624b74", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.34765625, -1.5703125, -3.84765625, -0.419921875], "student_probs": [0.6149657368659973, 0.09034158289432526, 0.009265094064176083, 0.28542760014533997], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "35987d0c24e2786abcce3f92255a492a99a095de01b2ddf468755bca1ded9342:action", "state_id": "ebe7ed94c8741e9ede8894b42da9cc14a3dc1fa906e2c83e9ee2c0b492695943", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0078125, -1.02734375, -3.06640625, 0.37109375], "student_probs": [0.3521825075149536, 0.12508496642112732, 0.01627989299595356, 0.5064526200294495], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f2072117a52b3e55c26f917ef7e8b86ed0b80c34d8684f57011283279ae15138:action", "state_id": "608349a9080d3590a42493e364449ffc0d31521717f5132e02896b2f681eba3d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.56671142578125, -1.01171875, -2.96484375, 0.23046875], "student_probs": [0.2531016767024994, 0.16219250857830048, 0.02300379052758217, 0.5617020130157471], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1c18805a2208ddc45a00138db8aa812ad488109d3f81aa236037eaa61f92c542:action", "state_id": "f7446b8394afe451fae244ff3b41b971e3006b4fbc06c4c73f78e85bad68fe5c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.40234375, -1.35546875, 1.162109375, 0.21484375], "student_probs": [0.12469913065433502, 0.04807579889893532, 0.5960693359375, 0.23115567862987518], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "25e75bb6c0c1d828021ebcd6bcd4148d9a9833ba318971b7c2e4e180caef5fe6:action", "state_id": "439c8a5a1812c387dbd16112023e4cb54c15c4b5cb071c88abd8d72dfe2a3c52", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.57464599609375, -2.05859375, 0.90234375, -0.4765625], "student_probs": [0.1490415781736374, 0.03379380702972412, 0.6527636051177979, 0.16440102458000183], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5da31f067d96b9ec7dc9692d6f183d8d7ad2206e5d80a882f4662c6f77efca7f:action", "state_id": "3e7634105aaf189cda8a98cd0c7afa81eace942c194ff224d449de755fef591b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5626220703125, -2.05078125, 0.8046875, -0.3330078125], "student_probs": [0.15603837370872498, 0.0352315790951252, 0.6124158501625061, 0.19631415605545044], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5ae7c2af6c77b4152e8e0a53fd0f6e11e9d721603b0568eb4648db3a01b93f5c:action", "state_id": "c6c23013e34e1a3a267aabb1d357db372aea926bdc0f7a1b589af4e9e2d932a7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.072265625, -1.85546875, 1.380859375, -0.560546875], "student_probs": [0.06779792904853821, 0.030979584902524948, 0.7881249189376831, 0.11309751123189926], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "800c11cb3fb94bf0b2bace01f12eda9654243790734ff2506a412b4a09ad34c4:action", "state_id": "81ffe361d03793930ec016d33cce06dbc7c7ba92648bd78c82a2e4c5487aeda5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7646484375, -1.9765625, -3.7734375, -0.28515625], "student_probs": [0.3375832438468933, 0.10047390311956406, 0.016660207882523537, 0.5452826619148254], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8ecb42723e730d612797849563b076a49f75899864d6aba4cd3d29537ed7ed29:action", "state_id": "1531eef050bfcb9b48058e27b7d85a3adf2cb34c8d8503913a7dff1d5fb2406a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.45703125, -6.4453125, 3.55078125, -1.30859375], "student_probs": [0.00659004645422101, 4.4926797272637486e-05, 0.9857205152511597, 0.007644587196409702], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b858d8a8679473a7ad54f853b40fce8855934992be2918549476ade441bd00ec:action", "state_id": "1ee8d6e6669fe1f94549223347f76dbe886cc5a6aa8f8b9b6caf2f3239b05c45", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.455078125, -6.21875, 2.95703125, -1.4765625], "student_probs": [0.01184406690299511, 0.00010107980779139325, 0.9764625430107117, 0.011592318303883076], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0b0840843db90722ee13892e3bbe31ab9cab5bc36e05c0110c49be6bdd450a2e:action", "state_id": "14540ffd51aff0efccc1ac890432d139c5b9bee780ecde6480bdfc729628a0e9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.07421875, -2.109375, 3.3125, 0.40234375], "student_probs": [0.030950212851166725, 0.004043956752866507, 0.9151597023010254, 0.04984620213508606], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a2972e6016f3ca1b9626c9a7c7a54eb53c3c5d0e5623500b04a95651677ac116:action", "state_id": "26550cbae37603cafdcb914c9e68aa0be4dd2710fc9017b0a789a46997c32be1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.296875, -1.0078125, 0.720703125, 0.908203125], "student_probs": [0.2154274731874466, 0.05843627080321312, 0.3291298449039459, 0.39700639247894287], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0ed1b5161ac1be99b05b3ec98588085dea8d88c1f6570853b0a4af7f7838cb50:action", "state_id": "b4e3e16f36303b496b9ce0d1861987bc082fb5f352b84dd29420d537240fe46e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.421875, -0.7685546875, -3.15234375, 0.421875], "student_probs": [0.24406376481056213, 0.17256082594394684, 0.015910204499959946, 0.5674652457237244], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "51075307193e299be39a4dbcab0808c3782830bf2cf066f5a024e6a53f3b26a7:action", "state_id": "26a81499e9aa433fc8980feae0f11df88c8f8c364d30262798731b3282ac4fd7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.14453125, -6.21875, 3.65234375, -0.998046875], "student_probs": [0.008110609836876392, 5.0739748985506594e-05, 0.9824486374855042, 0.00939011387526989], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "009bf7619bd783ea0ad86c4679473cf784368228bbbd6a18cfa313fa981f3624:action", "state_id": "83834c3c9f84a1eebba700fbe37c9148998f27d9acca308a581e686ab93c0e4f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -5.2109375, 3.11328125, -0.55712890625], "student_probs": [0.01524769701063633, 0.00023288461670745164, 0.9600703716278076, 0.024449173361063004], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a84d5bd5d4b1d1840ba45ae8b1a6453944c157a0edbbd98d5a55491e649c10df:action", "state_id": "71cc6be4cdbff0efe09283c2440640a678775f332bedd1d577323bcbaac457b9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.34765625, -2.21875, 3.611328125, 0.37109375], "student_probs": [0.01798240840435028, 0.002768484875559807, 0.9423515200614929, 0.036897506564855576], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "76aafb223434e3f1b248ca1e805bf7f5700719ef27cf0edfc134b0781eece0d8:action", "state_id": "fc7b0ae744e9c30e949b5b93486560e3602bc4e7db888f5fe18812ab9bb4b8bb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.68359375, -2.28125, 2.8984375, 0.58203125], "student_probs": [0.08996874839067459, 0.00463955570012331, 0.8241117000579834, 0.08127998560667038], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b9b8e44cb2c251b700222b951cb79672a2c8368f0c3d60e5533e923cabf418de:action", "state_id": "92eb8a32c101aaec433527016f2b25475e3f78e831d7182f55366fdfb16adf73", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.125, -0.6611328125, -2.484375, 0.5859375], "student_probs": [0.32105371356010437, 0.14627313613891602, 0.02362329699099064, 0.5090498328208923], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d640bb5926deee704680ddaa7f29f1e0a2a87f084f3efc9a34085dd980e3bbcf:action", "state_id": "b7c5bd05909e921f5453116fb1f11eddbc1f9eb490992b1ba7de9c811560ed54", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.314453125, -0.955078125, -2.3359375, 0.671875], "student_probs": [0.23037268221378326, 0.12139787524938583, 0.03051486611366272, 0.6177145838737488], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1d2074cf865a64f55c15c2d9c2394d8c1f29c9bf1658abc4b2403511292da0d8:action", "state_id": "4e6a79afb5a0196fbc96b9edfd1531a25b3cc51772fb1ebcfc7a340997849dbb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7998046875, -0.7880859375, -3.8828125, -0.05859375], "student_probs": [0.24061112105846405, 0.24344736337661743, 0.011025096289813519, 0.5049164295196533], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4fe811485a9057fe11ed3995aa6c9ef2acadbc9536c555804805a1abc91e04c1:action", "state_id": "226fd295be1b81ad371066231425ebd988528853400a2507f3b685601780fd8a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.4921875, -5.8046875, -7.9453125, -4.82421875], "student_probs": [0.7274864315986633, 0.07203090935945511, 0.008469490334391594, 0.1920131891965866], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "03638cde6c20b57ff45fb1596a2ba8ef22641cd8378734cb0cef494cdcc19662:action", "state_id": "0e338bdd72c638fd4ea04c8a6712d36d90ad3cb7deac4f7ff50cda058a26e19c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.359375, -4.3359375, -6.2265625, -3.7890625], "student_probs": [0.47980624437332153, 0.18069669604301453, 0.02728112041950226, 0.3122158944606781], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c079bd5027878e1bfd627540d168a4d3eb76c358783caf2f061ee7b06ddec037:action", "state_id": "03c59653fc28b3ea5def64e5ad0202b423c4873d2f3ed0588f4d47c161f8bf4b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5, -2.33203125, -4.921875, -1.63671875], "student_probs": [0.42734473943710327, 0.18596524000167847, 0.013953300192952156, 0.37273672223091125], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1a6993cfdcfabd772975947fa7970c86fda3b4d4eb9e46411f291093a35d4944:action", "state_id": "63839862637753d8f5cbebd354e4e97b9a303c552f4e291011248331e7868122", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.796875, -2.5703125, -5.1796875, -1.75], "student_probs": [0.3931795656681061, 0.18142256140708923, 0.013349166139960289, 0.41204866766929626], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b00f89d25d55f126093374b975f72753a452587e0fabc3710be194a55e47cb54:action", "state_id": "266a934e1e05473c07e70ba30b5936d554497144803b7af9399538c7f5d1cf79", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.02734375, -1.0, -2.90234375, 0.3671875], "student_probs": [0.35510125756263733, 0.1271108090877533, 0.01896728202700615, 0.4988207221031189], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4cfacdb2470e463086ff7c36b3d221e15668bf9727bcf82febf2fee92d17579e:action", "state_id": "a0ae93921b109ae81dbd58846b97577d6c90db579d1ec968398a4035bb72ce94", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3876953125, -0.970703125, -3.09375, 0.34375], "student_probs": [0.27004411816596985, 0.15074317157268524, 0.018038902431726456, 0.5611737966537476], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2e766eb47d7899a6caa3fc34617a90bb7ac032e73f721947f423326e379c5afb:action", "state_id": "8b7d3af788dd24eb9931eb0801e2b9de0f066dcbf2c1d1b2a9735ba843013d67", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.296875, -4.6953125, 2.66015625, -0.1171875], "student_probs": [0.04662024974822998, 0.0005732676945626736, 0.8970093131065369, 0.05579713359475136], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3adeed63472d796068cff5810cf3f19ba632fa75aadaf429d37e6723e986b831:action", "state_id": "61588853ea38b18739d8d4f674c9b730db79291d0a2b1d4483dfed1594ceec53", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.73486328125, -2.88671875, -4.484375, -0.09765625], "student_probs": [0.3299253284931183, 0.038359832018613815, 0.007762889843434095, 0.6239519715309143], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "86b8105c1750d0e21fef48484a098a75b0a9cf59c48779d88d489353e44dbd1e:action", "state_id": "efb68517c482990fd04b4009d75f1c66befc9e8f488e12eb0a14edd73f6c3740", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.30859375, -4.703125, 2.578125, -0.1328125], "student_probs": [0.04965517669916153, 0.0006129765533842146, 0.8905341029167175, 0.05919777229428291], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "382596546fc23efb22f44e8329a73183ed18e5559d6ef774b57d21128624ec78:action", "state_id": "a5494c2fc15528ce349104b81896a54861a5a11f8597b8d931916e0714d94bff", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6218109130859375, -2.90625, -4.00390625, -0.12109375], "student_probs": [0.35897472500801086, 0.03655480965971947, 0.01219659298658371, 0.5922738313674927], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2124bfa44c319d46eee6b1430b4203d6541c1d006d7e038618568d60baea19f7:action", "state_id": "cdcf87fbcfa46fddac79e8a8866bcfdc5f42e4ec5a2af0bc7b33c1753b6b28d6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.291015625, -4.3203125, 3.25, 0.44921875], "student_probs": [0.02658432349562645, 0.00047285089385695755, 0.9172107577323914, 0.055732086300849915], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e33a24e4d7f13d5a17c1da28bbb8d60a5fb74410719e058b0019ccde07f4152c:action", "state_id": "d401400245b87474005b3f1d12b11718520eea907755d1f031a051b156a31370", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.271484375, -2.828125, 2.98046875, 0.19921875], "student_probs": [0.035063792020082474, 0.002719718497246504, 0.9060750603675842, 0.05614132434129715], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b81d985669393ed564934de3dd1e37cefe9edf20f3a6737a28f42306f0fdf9c2:action", "state_id": "df13578a0505cf850c439186941894436f2f27a576e308053977ddde710c86e8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3525390625, -2.4375, 3.5625, 0.3671875], "student_probs": [0.018751448020339012, 0.0023310298565775156, 0.94040447473526, 0.03851306810975075], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "22855806274b8af9fad7e42ad1d62814fba7479b46b46e1749c05d1675ecb669:action", "state_id": "aaba7c9b35976dc1702e48c43c6c302759b6aa5c510fa5e7863df26db3f4b114", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.70703125, -2.37890625, 3.0546875, 0.6015625], "student_probs": [0.08060217648744583, 0.0036824867129325867, 0.8431812524795532, 0.07253411412239075], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ee47fe4c490762d8f18ab4d9221613dbbc83d8fef006891c2d129c4a7b8b0a52:action", "state_id": "607be3571e4729e90070256d7c0029f2088273bb58a51a47381b3d0d084db86b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.13671875, -0.63232421875, -2.5859375, 0.59375], "student_probs": [0.3216949999332428, 0.14909158647060394, 0.021135363727808, 0.5080780386924744], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "656b0013e7e7abd2b3e1dd523c224ddc1114919f151e10c69b12384712d12674:action", "state_id": "092f38149ba51ff4576f834da81a2ae57cd80501e5693fbf7ea3510dcae3609c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.98046875, -5.8984375, -8.0234375, -4.27734375], "student_probs": [0.7496911287307739, 0.04051582142710686, 0.00483892485499382, 0.2049541473388672], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "934645b5647f05cef9f09b90b3fed60cb21fae5b7c92a9b7c71aed530bab1e19:action", "state_id": "a7b0c53ff7455d407c89ebbcf578b2b4f8710da3bffd8b1f751cb37ab7872cf9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.49609375, -4.48828125, -6.6328125, -3.42578125], "student_probs": [0.6463919281959534, 0.0881657525897026, 0.010326230898499489, 0.2551160454750061], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a31ca6eb79305843a0e312ad12cce68bc069a4710d020b8f0a5cde96edf51bc5:action", "state_id": "d53a90e0f0f9d1da34bf3b3e41492e4f5ba199bb501936817160925810aae7ff", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.40234375, -2.5546875, -4.55859375, -0.51171875], "student_probs": [0.684991717338562, 0.03560106083750725, 0.004799296148121357, 0.27460789680480957], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "39f2e03f9ab63929572e8b967fff931be68e9d5781608f01645ba4b86689295f:action", "state_id": "d30a60171dcbac323bf13934bba7020b814f0f7a70c30b8b984fe22bd6c981bb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.107421875, -1.7265625, -3.34375, 0.49609375], "student_probs": [0.32616716623306274, 0.0646035447716713, 0.012820967473089695, 0.5964083671569824], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "78c5a9681ebd9e587b8b431a7fee91ceca10fe3a7057dcea9486b3b0a3427456:action", "state_id": "fb1df505703564cf32e7a16986dab536d5a48ac51117ebc33d3c2ca0635b8f4d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.25390625, -4.73828125, -6.90625, -0.12109375], "student_probs": [0.24163007736206055, 0.007411501370370388, 0.0008479481330141425, 0.7501104474067688], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6b2b9acb85050422dd3e3c55d116d7ee0df5e006f5828c8056a288cdf78d311e:action", "state_id": "6cb208d6b5c79211cbef7a2dff6d6673ee0a3584fc1e139638677509d21f3094", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4453125, -6.5703125, 3.515625, -1.30859375], "student_probs": [0.0069022648967802525, 4.104236722923815e-05, 0.98514324426651, 0.007913485169410706], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d4c107f36757c984cb1468af8b64e297ee3bd9706f721c8f13e0cf853dfa61e5:action", "state_id": "9e9d8bd0c41f4669511bd87ab8f139fda8c6d50183d1694a8e4128edb6c7ee24", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.564453125, -6.3125, 2.984375, -1.564453125], "student_probs": [0.010359448380768299, 8.980200800579041e-05, 0.9791913628578186, 0.010359448380768299], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "aebc2e43f66a1371a536d4e26be31e6c01054b4c4b73c1ccf11332d087eb33b9:action", "state_id": "169fd1ff0236d875bb050b33a17d17830558fd6d226591c2f1c7d4c3574dae98", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.14453125, -2.1640625, 3.208984375, 0.3203125], "student_probs": [0.03192073479294777, 0.004236445762217045, 0.9130324721336365, 0.050810325890779495], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "36d3dae8e70b719dbf3b52aa70c4a1e51f32e102d0f51eb681955319926b7661:action", "state_id": "d6b3846b23af74f7cee3a60b0175cdf4ee2c4663140a741361ad8dd11e337e28", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.140625, -0.5323486328125, -2.5, 1.0703125], "student_probs": [0.24300019443035126, 0.12397606670856476, 0.01732996664941311, 0.6156937479972839], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "38188507e156650146ba26de9e43346c015238cc4833ae73d50ad5a97b9f7195:action", "state_id": "9874a2f94a5e8547c11f5bcc081ad69ba34968b3d20d976bd1e0455388969fbf", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.578125, -6.1328125, -8.0703125, -4.8359375], "student_probs": [0.7282325029373169, 0.05659569054841995, 0.008153382688760757, 0.20701844990253448], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bf58c3af733365eeb3b00a675c31595eb143cc620ee4092b68187195eec6e07c:action", "state_id": "9cf1cc2cc3731184c95d54966b0bb3bbdf85b2ad02801cd7b5b23f958eb60731", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.45703125, -4.19921875, -6.4375, -3.12109375], "student_probs": [0.585284948348999, 0.10250496864318848, 0.01093129813671112, 0.30127888917922974], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "21c00cde2c13f244399db37c903d72da165f7de724b337a7ae97b7691b540a20:action", "state_id": "84d6d7174dcc5f6ab9e11627c621bb3a5bc4c957cf2102730b8cde100fd4a454", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.04296875, -2.36328125, -4.703125, -1.55078125], "student_probs": [0.7523879408836365, 0.07391675561666489, 0.007121338974684477, 0.16657398641109467], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "57ab27ec752dcba0e0406fe5bb4e8fab942298ef30bbfe6b3b9a0d37a547dd1b:action", "state_id": "379089abe53026afe77199553e3bcfe2b5c1b1c0f667e1b8419724eb4918ebdf", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.21484375, -1.63671875, -3.4375, 0.09765625], "student_probs": [0.3776508867740631, 0.09111251682043076, 0.015049035660922527, 0.5161875486373901], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "02a474a19aae0ad2f9f48979c316c970b636bd8eded5c51efc9aee22d1b1e96c:action", "state_id": "ae40de874f83ff422a1ad277335fb59e82a58e8997960bc77e5a103ae6de6a95", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.34765625, -5.6640625, -7.8828125, -4.60546875], "student_probs": [0.7175517082214355, 0.0707702487707138, 0.007695908658206463, 0.2039821594953537], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2d4a03e4ca0d5efee0a2fa7d3f226b4cebb32d11280592b6f35a524dca258f64:action", "state_id": "d4942b99b262dd01e02e15738eb8fecb5f9519d19ef4bee45836e55a1a4b56e9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.91015625, -4.1640625, -6.3203125, -3.2890625], "student_probs": [0.49924275279045105, 0.14247779548168182, 0.016493001952767372, 0.3417864739894867], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cc72483c6760a3c08008783c497b9f914e7fa47a939073dc2500482fdd0b4588:action", "state_id": "45ed1a18f2c8529ee842860a1ad8a6e869aa949918fe183d95c66c439d0ebf94", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.80859375, -2.78125, -4.8046875, -2.05859375], "student_probs": [0.4531324505805969, 0.1713191568851471, 0.022648433223366737, 0.3528999090194702], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d5833eb08c781f9ab38c1705e5865429f7dd103500e54197e5407981aadea25e:action", "state_id": "fe9a7bf03da4e601bb305760cc34a501c4a570524cbd53a64e7549329838f7bc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.38671875, -1.72265625, -3.78125, -0.3984375], "student_probs": [0.6278071403503418, 0.07616164535284042, 0.009720764122903347, 0.28631046414375305], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "91af299bae808bd087f3c923e54e22d791e4a02b660c5c2f48a14506c2d27860:action", "state_id": "3cc0f1bd3d011f834f4341aa4f440cbe895f3c9d7973bfe2d731b943f4c2f75c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0859375, -0.99609375, -2.91796875, 0.37890625], "student_probs": [0.36644798517227173, 0.12419157475233078, 0.01817324198782444, 0.4911872148513794], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d74391511af31286f0b68176f6823f1e9acfb9bca50d06eb6622e7b86820f30a:action", "state_id": "2c442b6c3cb598126277d37a7ca496371e4038c1b3dd2dc665dac2f796046ce3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.537841796875, -0.98046875, -3.3046875, 0.2578125], "student_probs": [0.25503066182136536, 0.1638181358575821, 0.01603122055530548, 0.5651199817657471], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a099ce4c3380edec2a56c24452a4831a0e75a4d06094237aa1c7d8c655972d2a:action", "state_id": "7e7fb79bb46549e5dfd64bb3b82a62fba19379b5b454bbb25a576ca9a4b16cb8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.30859375, -4.703125, 2.578125, -0.1328125], "student_probs": [0.04965517669916153, 0.0006129765533842146, 0.8905341029167175, 0.05919777229428291], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "736b4607bef8c704340feccdf8101ae10d034c118edd77ac99d7065e7461edd6:action", "state_id": "0ba4600a5daae1ff3b0b9790b688e815f0034dd3f9fc0f6e3c48ba0cbcc12ca3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6218109130859375, -2.90625, -4.00390625, -0.12109375], "student_probs": [0.35897472500801086, 0.03655480965971947, 0.01219659298658371, 0.5922738313674927], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6de79d22ca73532496485407d10707ca3e1ec8cb522d61a1d12d5ce5b275d15f:action", "state_id": "b166e83702ab06af446d090d73191a1be61f7e7c2499b1f3709fd69a52f7518d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.263671875, -4.6484375, 2.66015625, -0.11328125], "student_probs": [0.04810662940144539, 0.0005996880936436355, 0.8953799605369568, 0.0559137687087059], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "45dc1acf10fbbf447fa858bc1acbbb53a250e34c23c2c2f67032b5e2bca7c989:action", "state_id": "618841c9ecc1afc7dac45eecefdf3639792fed28650cd6721f0154f0435cfd42", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8095703125, -2.8828125, -4.765625, -0.09765625], "student_probs": [0.3141883909702301, 0.03951777517795563, 0.006013086065649986, 0.6402807831764221], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "82658d4a420556384cd3b9b2dd7501088ddd079f19cda861c2ac41d3d23b969b:action", "state_id": "6d77ecceb8b157d1be2ab4937f86d48186c764f68ba82892a26d8bbb591065ad", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1875, -6.09375, 3.62109375, -1.1328125], "student_probs": [0.008024215698242188, 5.938070171396248e-05, 0.9834411144256592, 0.008475261740386486], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c2ce13ed8fd6800097eec4f053017153e4ef6c6a08643cda2560a2fe051dfe11:action", "state_id": "5e74de44efe9ab6ca008f5b242481a337bb494d53becc081186b94387007f3af", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.19921875, -5.4921875, 3.09375, -1.06640625], "student_probs": [0.01327331829816103, 0.00018137057486455888, 0.9713866710662842, 0.015158604830503464], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "50d66366f5fdba233899e1f48cdd5acaeb4572117cea7561bf4c746c4868baea:action", "state_id": "c232ffbc59503f0f77ddc871ffd9b8750021739c726ac30f725b2164f5fca866", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.103515625, -2.19140625, 3.3203125, 0.31640625], "student_probs": [0.03000079095363617, 0.003718547523021698, 0.9206241965293884, 0.0456564836204052], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4201dc2dccb0e37e15b793def5f7f149efea2cf5f942954b6b470d125dc0bb18:action", "state_id": "49221281ae4af30e10df14ef8b4cf142b3ca94c4ef2f1208b5d02a01e7cde884", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.046875, -0.5670166015625, -2.79296875, 1.0390625], "student_probs": [0.2327311784029007, 0.12596352398395538, 0.013599597848951817, 0.6277057528495789], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "388f5187fa587b96331ed9833c38dac6cb3b72a8cf087f4a267275041a91872a:action", "state_id": "60c4574179b36a7e9272575acab886016086a423d31e3ed0ddc944c5aa18fc65", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1796875, -6.140625, -8.0390625, -3.9609375], "student_probs": [0.840106189250946, 0.016000032424926758, 0.0023968450259417295, 0.14149697124958038], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fd9bbba5cf6b06c99e24bab0ab8a165ce638102ae63fa1285c145d56882d5dc9:action", "state_id": "aa5b4ce1300d19a8c480367ce2759c3aa86eab5ed7d94f8beb1104de1430f106", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.4609375, -4.203125, -6.5625, -2.50390625], "student_probs": [0.9417195320129395, 0.008878610096871853, 0.0008388444548472762, 0.04856308922171593], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "202bb68745018231ea9252a2b31c2765622c6b137543e83a67110175190a7d79:action", "state_id": "d478aba9fb03d255b01bb7c67bce06ab32f046575ed98f19a4387214547158b9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.23046875, -1.90625, -2.734375, -0.4189453125], "student_probs": [0.5910298824310303, 0.0697660744190216, 0.030478540807962418, 0.3087254762649536], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a59db1f19b0c92c2145f1dc72a670641b59221e94cc87b8173f262c59300b11b:action", "state_id": "82979e08a7fe8de849cb0288464f2764c17a644bba6a85163b9b2e82f9622107", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.01171875, -1.15625, -2.9140625, 0.671875], "student_probs": [0.2981290817260742, 0.09491629153490067, 0.016365619376301765, 0.5905889868736267], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c6b0bd9877fcb3f6a40b1249247e8db9f72463fae32338738d824a483762375d:action", "state_id": "dff17173973ddebe47b2e40843dbc1924b0d98234612a971aa661942466222e9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3212890625, -0.744140625, -2.546875, 0.37109375], "student_probs": [0.26584064960479736, 0.17417238652706146, 0.028711887076497078, 0.5312750935554504], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "aec376f668e450c3d9ab666688ed5cfcd4b492c231bc13f068aa972253d386c6:action", "state_id": "f3c2e7ed95fd56046883c760ec43f0f430d9dcbddfa38de46ea0b3354a46f915", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.52880859375, -1.48046875, -2.86328125, 0.1640625], "student_probs": [0.28715750575065613, 0.11087138205766678, 0.027814524248242378, 0.5741565227508545], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8604ddba856107597ad1204006cce8fd1cae342b0643b0652004c16207661e13:action", "state_id": "85e6009fdd8f80c557238dc3ee4db7e7e216b11b258d47d97285770d8efa6b0f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5634765625, -1.8125, 1.830078125, -0.109375], "student_probs": [0.0723908543586731, 0.020760590210556984, 0.7928504347801208, 0.11399806290864944], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "86e3df58553bb7c86470c89624fda782b99d7620adba4de266df76e722a4cd4c:action", "state_id": "1f1c1066f7d94837f64abe755c46154f71dddf2ab1b5f045d71d4d96267dae7a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.818359375, -2.09375, 2.19921875, -0.55908203125], "student_probs": [0.043446071445941925, 0.012135437689721584, 0.8881126642227173, 0.056305814534425735], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fac251e4039f8c145f0201841d9b20119a028cd6fa468a63b1a49a4ecf3b8d3e:action", "state_id": "4bcbaaf0a3670447991abb9d0615b7c2199c1bfe6fb79ddd821cdc1fafd2dfdf", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.64306640625, -2.01171875, 2.23046875, -0.3671875], "student_probs": [0.04933005943894386, 0.012552015483379364, 0.8731163144111633, 0.06500164419412613], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "832bd8e87a121c5651637a3ca3ab84eaca476b2984d0ef925ab6e15987735694:action", "state_id": "a9ca146f56e84ecbd58ca18baad38472792deab5d08997c7904ad34bf6f38a6a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.076171875, -1.79296875, 1.7890625, -0.9111328125], "student_probs": [0.0494537316262722, 0.02414894476532936, 0.868069589138031, 0.05832767114043236], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "08af039707d3f5ac5541328ace6b802ab115afa26a0d460600bd4edc69bd72d2:action", "state_id": "fa51367245c14b57a3639c3479dd6c975aae6bb074107fb4205b894e1d245cf4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.76171875, -2.03515625, -4.04296875, 0.07421875], "student_probs": [0.2759058475494385, 0.07721719890832901, 0.010368886403739452, 0.6365081071853638], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4187b8f4bbd50ebb57ad728a8acb3698e90cbcaab093b4c7ba9db93aedda8f2f:action", "state_id": "ed57c60e13837a88e76a3656fc60e1d22c13af84fa53bdbd66d7819ccffbd5e1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.37890625, -1.4375, -3.484375, -0.205078125], "student_probs": [0.3873569965362549, 0.1343909353017807, 0.017354954034090042, 0.4608971178531647], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4c32d68ef5246a49f24076b72ebc0b84ab72d00092b56e0f555713017f340a79:action", "state_id": "985008c38fa9d1387e9f8360e198b4c503243570ad28e3d7f82dd93b972afed0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.06640625, -1.8125, -3.46484375, -0.4609375], "student_probs": [0.5313847661018372, 0.0927022397518158, 0.017761779949069023, 0.3581511676311493], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a206c4ac3995b330e57fa4947a840fa035a0ba28c4d4cbea7680b4a6d7f4298e:action", "state_id": "a846a48928071c1bf2a0a68e07ccc3a63098e6f491dd4bff8fdb2147daaecce9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4111328125, -1.47265625, -3.671875, -0.07421875], "student_probs": [0.35907840728759766, 0.12421543151140213, 0.013774218037724495, 0.5029319524765015], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "818fabf7a5c905a061d87345f444bfd67112dec38cd7e5a6a61ee6756b5bbd17:action", "state_id": "0722c9cabf243f328da6e6539b4b0ec99c6c147a7dc5d001d5d5f97fc46d441c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0078125, -6.0234375, -7.953125, -3.85546875], "student_probs": [0.8487120270729065, 0.015303704887628555, 0.0022219994571059942, 0.1337622106075287], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f3de78d90ea06c5596b645a81370ec388b883d73ac2e8d7849aeeb734097451a:action", "state_id": "7cf46d3aec7f4a9e8756dca72e4d70e0cabc7e4362756ecfb8298354d8516e57", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.71875, -4.203125, -6.5546875, -2.45703125], "student_probs": [0.9526162147521973, 0.006940245628356934, 0.0006608520052395761, 0.039782650768756866], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "58c5e15af078950574c08c652d4185d8b0d87427414cb457dc924e86954a8bd6:action", "state_id": "4e5fbffcb5c8f8d20e6d83c76efcd6d8659b0de0938ee46d2cc4c0a749c39234", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.37890625, -2.328125, -4.0859375, 0.07421875], "student_probs": [0.36494266986846924, 0.051962461322546005, 0.0089594516903162, 0.574135422706604], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6c5c51ab8da879d4613ddc591e60d2be4103794ccb0f1946d9a016de48646d79:action", "state_id": "86aa303088f56f686c1c65fc1dcae324158d554d30754a10ae07312d39b3a86d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0546875, -6.25390625, 3.6796875, -1.005859375], "student_probs": [0.008632008917629719, 4.765625635627657e-05, 0.9822563529014587, 0.009063953533768654], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9fff73b2192cf97302362baea6c0b86610ec3f578dcc66244f691bf138c73979:action", "state_id": "a193309a299952f86beb58b4480348a6493d133fc15a50a7e979edbb13cabb6e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.33984375, -6.0859375, 3.0859375, -1.05859375], "student_probs": [0.011639879085123539, 0.00010109882714459673, 0.9728386998176575, 0.015420334413647652], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bf4749d7b15f40268a023909e62889964bdab665357712edf12187563fb4fb67:action", "state_id": "842e543b3f1cc800a72cf9945c90d042e31ec8ee2b69891bbb1293e51ad9a1d2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.185546875, -1.8046875, 3.705078125, 0.40625], "student_probs": [0.019250478595495224, 0.0038129196036607027, 0.9421465396881104, 0.03479009494185448], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "349ba595cf8341bdb5901117e48882b6155b163da6183c90017abaf26291942f:action", "state_id": "1268ee4e5649159608105eba645d47c60dbb23d5becc5737712af655ef0c94bf", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.25, -2.02734375, -4.09765625, 0.671875], "student_probs": [0.2699480652809143, 0.04564462602138519, 0.0057579027488827705, 0.678649365901947], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b07fe09c7b47a6a09de21b91d6bf920d67592374ca12ea38d4578cf2d7b03b70:action", "state_id": "4ff3283cc104e678b6a2d82f032633bf51643249833d73d8f0c1db52e8bcef2e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0546875, -5.9921875, -7.9453125, -3.80859375], "student_probs": [0.8365690112113953, 0.016310498118400574, 0.0023133207578212023, 0.1448071300983429], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6010f0cf3e2f95964798b003c042c122529a8415ad86ab86efdc3e06890bae37:action", "state_id": "b919fe29e776b20331003b7d52e6c50c9bb538887879062a2cc69e0c5eb56062", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.4453125, -4.203125, -6.5546875, -2.625], "student_probs": [0.9461677670478821, 0.009061026386916637, 0.0008627933566458523, 0.04390847682952881], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "08f90a230a5b86f43f8e8f6c62ece74a10cc146b4e4b7ca5391affa2d7817ba1:action", "state_id": "1aeff5888c2dbeeb5bd75e71567dafe54edcc95be4fea4376e0c315ac09dcb91", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.13671875, -2.15234375, -3.21484375, -0.798828125], "student_probs": [0.837155818939209, 0.03121653012931347, 0.0107881436124444, 0.12083952128887177], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f88d94805ff6f3cdee336fb10b6b50bc85d112d14e74177b5c3d05a7fedf8c4d:action", "state_id": "9c1ce6a06f589d49551bc7670994e021db1e44d9c63e62a16bdf989e8dcca31f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0, -1.578125, -2.87890625, -1.02734375], "student_probs": [0.37219926714897156, 0.20878486335277557, 0.056856077164411545, 0.3621598184108734], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "62d262c3f8569c71274ddf1726a0f067bb532ddd2bd6700961d08a807c8f8170:action", "state_id": "07b851852318fa7d6e22aab25438dab23e7b59c99689ceec51a566ca31a6823f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.7109375, -0.873046875, -2.5078125, 0.7109375], "student_probs": [0.4454023540019989, 0.09137699007987976, 0.017818335443735123, 0.4454023540019989], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4a26c696dcc41e635b92e7bcc823ac516b9cd8e62d91d07123eab3379999b309:action", "state_id": "d95419649cd97991164390ff22295041c9c060de7c283ab6966dafd1828e4609", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.653076171875, -1.5859375, -3.54296875, -0.322265625], "student_probs": [0.35197576880455017, 0.13847656548023224, 0.0195635836571455, 0.48998409509658813], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bd29359fb0a56efa8a04c4236773c8191d556633f3f18de8c7e07d8cc9d9c342:action", "state_id": "c1252662f5acd359336b4c82f0338ea22c8450d5b6a447ce54010ed725b45b53", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.779296875, -1.91015625, -3.90625, -0.34375], "student_probs": [0.343357652425766, 0.11082065850496292, 0.015056646429002285, 0.5307650566101074], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "44e9387d04cb461fb28086a14de6a59eaaf5b151547c8ca38e81bdd53048a9bd:action", "state_id": "0912de5262436201174fac3f67d467af04d30505a5bcf8e6258b4cd7cd6456e6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7890625, -1.6640625, -3.16015625, -0.171875], "student_probs": [0.2972697913646698, 0.1239204853773117, 0.02775861881673336, 0.551051139831543], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7039853a64c5afbf36702413a852545fc7500dd7c7b9212014e6e2880e2b298b:action", "state_id": "54b0337afe940dbb478a3e59f98831c83783f6d2a154af2748387a9d3e716196", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8701171875, -1.7265625, 0.046875, -0.52734375], "student_probs": [0.18743182718753815, 0.07959648221731186, 0.4689083993434906, 0.2640632688999176], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "35d1f54fe77953f25b249d3e20f57dadd2c33858ed6a3248465d5365b279a835:action", "state_id": "9ed7e219f10845be7a4553f0e8807791049623d44a7cbaa66a62417dae24614f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7890625, -1.42578125, -3.390625, -0.4716796875], "student_probs": [0.33594122529029846, 0.17772145569324493, 0.024912599474191666, 0.4614247679710388], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d712553cd6a3e39a755a3ba4fd1ded12ce1ed7af5f3b758bb390c1ae16da99dd:action", "state_id": "b03b18dd31b70666be8da409935dae75e2b07c63615a1b6b03c62b51a825ba7f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4609375, -6.52734375, 3.578125, -1.30859375], "student_probs": [0.006389934103935957, 4.028877810924314e-05, 0.9861283302307129, 0.007441465277224779], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "091ae4b6500ed2def3c699864fc6bf65001ac08237a91acf71fb753cef874fad:action", "state_id": "dc4ed16a84cd3859dcfc257131ab5dc3a88336a64390b551e9d3ee187cf98164", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.380859375, -6.0703125, 3.00390625, -1.33203125], "student_probs": [0.012153820134699345, 0.00011171441292390227, 0.9749724268913269, 0.012761996127665043], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ed897d7ab8a317608b1fd9f753482eade5df2c7b030206febb441b2d3dbc7930:action", "state_id": "f92a6560f0ad41629796d2213541e3b779c5c75dad35644cd025a41f04aa444c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.125, -2.25390625, 3.04296875, 0.30859375], "student_probs": [0.03784877434372902, 0.004502768162637949, 0.899255633354187, 0.05839278921484947], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "21776c597346105821421d69382de7f14031b318ebed6fba7fbc87a432797d40:action", "state_id": "5d71a54bc5484554054f19556ee1dd32ba1363df0f4ae63e4f5e94d52c111788", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.25390625, -4.625, 2.66015625, -0.1171875], "student_probs": [0.04856569319963455, 0.0006137446034699678, 0.895139753818512, 0.05568084120750427], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5f1fcd44078f987192bba3ee6277aef831bb155dbc97a79e6ad3ecd90d0a6aac:action", "state_id": "144cdf3d23f8fb2899011919d277f84bbf9d831ac9625bbc14564f6a36324243", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7900390625, -2.8671875, -4.7421875, -0.08203125], "student_probs": [0.3150158226490021, 0.039467379450798035, 0.006052518263459206, 0.639464259147644], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d6f33566825f7d7d75d7ceac73b1ea7e03de287641b2af75d6cddc768d07d68a:action", "state_id": "7c2a94b621e539040d212a0a578aa11b70ea410e14ec6cb7a9d8749988f6a6e1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.982421875, -6.125, 3.68359375, -0.9150390625], "student_probs": [0.00922943465411663, 5.392395905801095e-05, 0.9808439612388611, 0.009872771799564362], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7519695e30c78c2d17f2d89d13956073f8c902611dee4836f7d9f71954efadb4:action", "state_id": "7b78f92a6d8809b09c213bb9a3162da48877162fd396021a8872bf098d7d763e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3359375, -5.921875, 3.18359375, -0.98046875], "student_probs": [0.010612376034259796, 0.00010818456212291494, 0.9741371870040894, 0.015142261981964111], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d7a054e6f7e228b5d349c108ff5f279717f2c661b7e44e535f89ad32c97518c1:action", "state_id": "01af21274ad63eb5319c4a7532ac98b0e6a3beb1c8223f366fd1ef6b202576db", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.212890625, -2.01171875, 3.748046875, 0.453125], "student_probs": [0.017979631200432777, 0.002975498093292117, 0.9440480470657349, 0.034996747970581055], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cb286c100ce399a51816ece798ee1f89fc85bf5c740c44eca2433877c1012328:action", "state_id": "3dac477f01184cda0e329a77d32500a43da5b74ca00977482d3cace573c5f127", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.67578125, -2.2734375, 2.8359375, 0.54296875], "student_probs": [0.0943351536989212, 0.0049413335509598255, 0.8181209564208984, 0.08260262757539749], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bef53f1a1e55b7b1dc9500caaff41af14f09759fc3e3b8f2ac1dac36326db79c:action", "state_id": "75a95d548ada5591fd46c53832bc45992e19ceed7e5c65840194670867977f7e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.03515625, -0.6611328125, -2.6953125, 0.52734375], "student_probs": [0.3125477433204651, 0.15578365325927734, 0.020374590530991554, 0.5112940073013306], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c478ec87ff99203b3f02b203f52023f6f2d33a0bcaadb772dfdc4f51742aa554:action", "state_id": "59a6f60638b7e139abae104b3c3fd556b9786d2a9e23294c05f25a890c340b3f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.125, -0.384765625, -2.58203125, 0.23046875], "student_probs": [0.30452799797058105, 0.234861820936203, 0.02609468810260296, 0.4345155656337738], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d144f91736b2bb26fa407b0ac7528b35b1403da53a387f7f8181ba5a11950b18:action", "state_id": "78d6b6d36d1dc0051b303dee39ae5854d716d9b8cb2b566f86f587f11718477d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8076171875, -0.9453125, -3.86328125, 0.03125], "student_probs": [0.23628373444080353, 0.20588918030261993, 0.011126941069960594, 0.5467001795768738], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ab1dce7c648f5e4deb2e9327985fbab66cc140ed875f690f45ebb2b0773f41ad:action", "state_id": "9e681fb4fa3fc7bcda7fab448efcda391a9097bc8ac3d3b550aedeec87cd4b7c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.02734375, -5.01953125, -6.9375, -1.171875], "student_probs": [0.7640566229820251, 0.004912421107292175, 0.0007216595113277435, 0.23030930757522583], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bebc790db4893bd23cd7bcf76a6a41c57df5f1e5638a44520571da235bd099a7:action", "state_id": "6733084784f9f5bbbcf8756dc6af77ce76e3f080ea9b2d5229dc84462d1438be", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6455078125, -3.34765625, -5.1015625, -0.2421875], "student_probs": [0.38828083872795105, 0.02603861130774021, 0.004507191479206085, 0.5811734199523926], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c3e7d0ac1e77617d47e1482dc4b2335c36545dac104d91c0b959a0d8260687fe:action", "state_id": "92d66c752475f785bf62436aba62356afb60dbc9e9b161570113123ede78b7f1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.54052734375, -2.30078125, -4.11328125, 0.109375], "student_probs": [0.3209826350212097, 0.05520939826965332, 0.009012686088681221, 0.6147953271865845], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "189767d0a1ac2115a6a9e2499a0c26186a966ef9f5c22a9eaa1abcd18fb446b4:action", "state_id": "fed58948dd16155df9bdcdd7b44620959c0acf15978973903e748205def3c82d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.166015625, -1.73828125, -2.8203125, 0.51953125], "student_probs": [0.306487113237381, 0.06361886858940125, 0.02156084217131138, 0.6083331108093262], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "774f213980c90da34070b6cd713cdb36e8c5c5080e688ba66b7a41945b451351:action", "state_id": "9b24629f92d6fb56ada99a2c0de9796ef4379eda45d73daee1806badc2785d26", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.01171875, -1.61328125, 1.369140625, 0.37890625], "student_probs": [0.15321581065654755, 0.030169982463121414, 0.5954213738441467, 0.2211928516626358], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1a23d28add601d60575f45c83dcaf3cdb25f88b8a32519a8483ab29731c116c5:action", "state_id": "d4eeeeba2df62bb0431d7244e524ef3f75582ecaf0d96d63da71d87ebf02e972", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.111328125, -1.453125, 1.419921875, 0.359375], "student_probs": [0.1335746943950653, 0.034913163632154465, 0.6176431179046631, 0.21386905014514923], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "16759b6e7eefd970e7a7ed55079f41f7c4344dd1925063da8dcb0909fecb5796:action", "state_id": "97a15e2d5bb2c0b6f4bc3079bcf5e55f30add7d0cbcd492ee71ef43f9be5ea5e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.633056640625, -1.34765625, -3.6484375, 0.15625], "student_probs": [0.2673593759536743, 0.13084246218204498, 0.013107869774103165, 0.5886903405189514], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5923870a2f81aa72b527b8250bd55dc338d42d58453150837918446a2d3fee4c:action", "state_id": "39c27bbf1b2687248b0468f5838b1a51be2a09dcb81d82f4c424d61eee5ac4b7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.0546875, -5.0625, -7.0078125, -1.25], "student_probs": [0.7632128000259399, 0.005102468654513359, 0.0007293598027899861, 0.2309553623199463], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "12f293d1355919babfd267aceeb8af13db8be901f87b1014e25fc1477d3968fa:action", "state_id": "ec4fbfa202e076da2bed7cd97924d1da770ea97439b5d49fcd0cd51745477cea", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8544921875, -3.72265625, -5.46875, -0.4296875], "student_probs": [0.3852073550224304, 0.0218809787184, 0.003817225806415081, 0.5890944004058838], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8713788ed7a434f8cc1b6e9541249dcaa1dd21b79d3a8e8df09362c60c6436ee:action", "state_id": "5d19cdddc87a6f7c64b27c2b3d74a11a187a8eaeaf1bc0bff487b15578a24aa2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.49755859375, -2.27734375, -4.03515625, 0.109375], "student_probs": [0.329755038022995, 0.055621225386857986, 0.009590302594006062, 0.6050333976745605], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d010e860e3d49192271b69b2e8498f3c57042884664b8b5140314a5f62d2d620:action", "state_id": "80ea04a9254654361dcc64a219a50cc3e93ffe945bb41322230bab83a7e18242", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.099609375, -1.7578125, -2.84375, 0.54296875], "student_probs": [0.31684061884880066, 0.06035210192203522, 0.02037397399544716, 0.6024333238601685], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "87fe325647a1da7757617b28a2799be79564e4e6b064b971c751919788284ed2:action", "state_id": "ce6458b5931a000a632c57f0239ec04b81e4a18a348b60d681bccd417d15def3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.171875, -1.32421875, 2.064453125, 0.4375], "student_probs": [0.07990998774766922, 0.025243207812309265, 0.7478698492050171, 0.14697696268558502], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5837a3d5e16b955416db50e15606b55738d4334fb9ea3a1ce01011bf759bb32f:action", "state_id": "2bf604f7845b9f568c270837a8d3aca5941f64ff98bfd618370ad970cb93bf0d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.0078125, -1.47265625, 0.87109375, 0.5390625], "student_probs": [0.1863160878419876, 0.04306027293205261, 0.44869837164878845, 0.3219253420829773], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "54b209b0b18f8b66242eb92e585caac5b3a0d7715d7288174d36936771a72720:action", "state_id": "04b46cf27e75f7318c67c836841c814f84aab5249e63ccfea93f8100340ef6c2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.62646484375, -5.3984375, -7.4140625, -1.79296875], "student_probs": [0.756976306438446, 0.00640679569914937, 0.0008536229724995792, 0.23576322197914124], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f1486ca6700d0e01408d6e8597d72ff1b1adbd900f211892cad1e8237a20e921:action", "state_id": "f823ad38a843a762029d0fae2502df30d7b6a6da1e5596dbe15361f6df686d83", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.36328125, -3.96875, -6.015625, -1.134765625], "student_probs": [0.42732471227645874, 0.0315658375620842, 0.00407634349539876, 0.5370331406593323], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "36215e69e792b5fdf10b10bec639ae2cd0ebd81cd44346f2182d3ebbde4fc1f1:action", "state_id": "2d2825754e3e4b94700e386d91cc81a50ff05a17181be3d821284366470bd804", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.328125, -2.8046875, -4.875, -0.15625], "student_probs": [0.222951740026474, 0.05092698335647583, 0.00642425287514925, 0.7196970582008362], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0f383b93a7fb6e1af675039314e78be558fe1bf12abddb3dde2d4b89b03c45b2:action", "state_id": "1c625c858d4bc70958818ef7562eb4e314b9fa473b07da31e5e96aa56898fc42", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.43359375, -6.1796875, -8.046875, -4.1015625], "student_probs": [0.8224436640739441, 0.019417723640799522, 0.0030011595226824284, 0.15513741970062256], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "41932e374e5205f40f5edb5ae91f252064f7675468dcb31f4c8a2bd8c4cc9b56:action", "state_id": "4f9d6b1f86b2d2c80103d1bf41c2c6f0e95f22c1d2e59c84783279ecc56d242e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.38671875, -4.44140625, -6.671875, -2.8046875], "student_probs": [0.95240318775177, 0.007620663847774267, 0.0008190539665520191, 0.039157118648290634], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "42249eb556cb37b2a83fbb70e2e2a7ddfa64198a400be06fda7f36bee9bb56ae:action", "state_id": "7365123d19eeb7194deb1b7caa579807ad98577c9db8e3d17d5c918cbe92c2da", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.646728515625, -2.578125, -4.421875, -0.17578125], "student_probs": [0.36109039187431335, 0.05233847722411156, 0.00828114990144968, 0.5782900452613831], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "10bbcfbedac73b71c3788a88482f59f8acafeea16e84f95d6c3127d72082008c:action", "state_id": "8fe6b474f17bb0be432873de2a6407289df2c29d1299310e56fde2c3d80c698f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.43359375, -6.1796875, -8.046875, -4.1015625], "student_probs": [0.8224436640739441, 0.019417723640799522, 0.0030011595226824284, 0.15513741970062256], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3bce86af5a0100bc2f6c1f53a91482d8f6ff9b9744df3cf901a334431ad7e73f:action", "state_id": "9ea8cdbc130661ced9969ae184d70812b9b94de89803481add8c6e3a07117513", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.6171875, -4.41015625, -6.703125, -2.74609375], "student_probs": [0.9598406553268433, 0.006292909383773804, 0.0006353716598823667, 0.03323109447956085], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "57ac2d4a4daf7542f5ee7fd9c04bf8185ea52ffe173aa8af157fcbb1a2238a6e:action", "state_id": "cc5ce9fa0e47bcb8b6917ece92983cd65e97820fb3fd26d020d6645a13d74bfa", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6232452392578125, -2.53125, -4.3828125, -0.078125], "student_probs": [0.34524595737457275, 0.05122626572847366, 0.008042097091674805, 0.5954856872558594], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0de196fd6cd30d0ebe77925c053f123e53630c4ceb50e9f7e3e44df5bf043339:action", "state_id": "ad2bdde7ebe812fedfe34e7713179cc27fbe311975c0e851f2a5ff35d7455032", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.43359375, -6.1796875, -8.046875, -4.1015625], "student_probs": [0.8224436640739441, 0.019417723640799522, 0.0030011595226824284, 0.15513741970062256], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6ab433cca155fbf902a5e00dca0a540a73694ef6f3ad231f01f0a1c3272110a9:action", "state_id": "6c0a8b57be6c00984579b8eb66254075a796d7dc6e9085c3409427cc47befb20", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.6171875, -4.41015625, -6.703125, -2.74609375], "student_probs": [0.9598406553268433, 0.006292909383773804, 0.0006353716598823667, 0.03323109447956085], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7254f2ba86588376dffbd276fc44d6448ecc579becd4502f76cc3bbeb7399a61:action", "state_id": "74bc783d90c19416d8ce699413b87c8f301e0148d91f7e6d28a55bd5c97a501b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6232452392578125, -2.53125, -4.3828125, -0.078125], "student_probs": [0.34524595737457275, 0.05122626572847366, 0.008042097091674805, 0.5954856872558594], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "74620e928ac624721687c7b4842c2973a7884194ca58b656dd4e6732cb08f36b:action", "state_id": "0aedcdeee8741c8d6d8a11a6bb335747e03fce58f4a937165aff34ebd93c49e8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6015625, -6.40234375, 3.45703125, -1.24609375], "student_probs": [0.00625766022130847, 5.1458740927046165e-05, 0.9847620725631714, 0.008928737603127956], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9e8e0c44be32367eb26eb8e07a76066bbe870ba0d3cc3ab0d8ced0c4abac40fe:action", "state_id": "b2ca0292bc3a0987533ff3038fbdae71ce116dda6cf52e07d92170433a55950a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.064453125, -4.72265625, 2.96484375, -0.8544921875], "student_probs": [0.017099637538194656, 0.00044080798397772014, 0.9613649249076843, 0.02109462209045887], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c6a8d95833cda6cfd4349419d2df5174882249d6a4b16b12bd054f8096ae20af:action", "state_id": "a9a9cd86845d509d872b4007cda04fb0f86c9afae819d9d9406f68a97ecd422d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.15625, -2.19140625, 1.55078125, 0.1796875], "student_probs": [0.12433969229459763, 0.016246233135461807, 0.6854314804077148, 0.17398251593112946], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4bab586cc169638583b087ca5bda81eb9a6c99bcf0100d97d2752bf28223845f:action", "state_id": "69ab6003d2a3c60a1373ac2bd00f526d880930eac01a2a97de39f47a745e2832", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.212890625, -0.4482421875, -2.70703125, 0.671875], "student_probs": [0.23281387984752655, 0.18399116396903992, 0.01922283135354519, 0.5639721751213074], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8a4a3790767817b570625af5fcd2b81315704301376ca40cdb0e9d42c6c4ffe7:action", "state_id": "6ebe99cde810fd150b20fc22292a9a5ff9f402416f30802dd3aa9fa0febe68d4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01953125, -6.16015625, 3.65234375, -0.978515625], "student_probs": [0.009178884327411652, 5.37334599357564e-05, 0.9812042117118835, 0.009563188999891281], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c0bb5cf9ace09f2a083fba4e472adc029667c9ea4fa1e23f8c2204d9955cd79f:action", "state_id": "6b4d8b9369159c0416c263b8ae030c1bccef0941acb44a9a2809516d9d947d09", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.73291015625, -5.0546875, 3.1171875, 0.140625], "student_probs": [0.019838793203234673, 0.0002633850963320583, 0.9323766827583313, 0.04752112552523613], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "31787bdd27b536874ac0c6c4d869529d1ad15ccccb648f63b826c9cdaaf4ddbd:action", "state_id": "c5e68c8c77d4cc8953553a62d3614d026c251e1a56c67b5987eb8ce9e4d50953", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.65625, -4.2109375, 3.16015625, -0.525390625], "student_probs": [0.02100442908704281, 0.0006005230825394392, 0.9544540643692017, 0.023941002786159515], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c6ce350cc6d4aa07a297ad3ece3df0749b2fb470b4d0670d9a667dba35887c7e:action", "state_id": "a55dab693046683b81868ac19582bccb1a321046cd2f2ffddfdca4cf0b5bea70", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.087890625, -2.18359375, 3.26171875, 0.67578125], "student_probs": [0.0314854271709919, 0.0038721957243978977, 0.8970702290534973, 0.0675721988081932], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8e1a7f8b210a3b2bbe7afd3382214080ce7800d8c313bf1493e493307e8bf44d:action", "state_id": "7cb8cb1494ffddcaf78690d9e2a8c40ba4c81d12b52c0659f9d716aacbced57f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.31640625, -1.4765625, 3.486328125, 0.94140625], "student_probs": [0.0372573584318161, 0.006202055141329765, 0.8869346976280212, 0.06960590928792953], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1e5e18fed0e6867cfecbe562be210fae567f7e78c7c59628cacb42142f584be8:action", "state_id": "cab362f7728681b0a13722fab0af4af0fe70589521ad4156acbdf4c28307e9c7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4208984375, -1.609375, 2.744140625, 0.203125], "student_probs": [0.03722900524735451, 0.011343121528625488, 0.8819428086280823, 0.06948504596948624], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "395b6aaf35c3d0180a5d34656e87962ccdb3cb820111028c4eb160c6fd66686b:action", "state_id": "ed517e2d0cbcf83b0e2be668533c72c1232dcc9563bef7dda4e92cb60a56fcfc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.43701171875, -2.08203125, 3.330078125, 0.24609375], "student_probs": [0.021539175882935524, 0.00415725028142333, 0.9316556453704834, 0.04264793545007706], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "45c43ee10b08ca5387a0c393e827ecaf3d707a14ab51a4a903f1ec43cbf929bb:action", "state_id": "ff37c25736238180841ad0d1a67a6cc38f601c736c4e5d46c49e379885a76609", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7802734375, -2.171875, 2.935546875, -0.458984375], "student_probs": [0.02287289872765541, 0.005687957163900137, 0.9398996233940125, 0.03153953328728676], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e7f4bb91e5f064f6365ca3ebabae4a53475ab088b750624d85af31fe233dd533:action", "state_id": "2028d4093bd35797f5f87bf5c6dcb0dce95b3d838708f3676466611aef730943", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.1953125, -1.9375, 1.71484375, -0.291015625], "student_probs": [0.11314758658409119, 0.019816314801573753, 0.7642151117324829, 0.10282103717327118], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "081ada2e930adf0c8553b78f1599062b7d3e3f8050c6eec35214effbeaf541c6:action", "state_id": "4cb6908ee903af00fab01d37f909217ecb76829a3e84a26ea6f20a3235f5d253", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6220703125, -0.982421875, -3.53125, 0.03515625], "student_probs": [0.27163687348365784, 0.1894480139017105, 0.014809761196374893, 0.5241053700447083], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2ad2e8895730e00ba6e027c34ca566b027ffa725f01335b5931f26e19cec604d:action", "state_id": "85aab267414c2ca71ea07ee685c79390a92fa210cc65b0ab60cc8c36d153fd22", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.67333984375, -1.25390625, -3.0546875, 0.0390625], "student_probs": [0.2709372937679291, 0.15161144733428955, 0.02504163421690464, 0.5524095892906189], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ce88fc2b075d4de648b131c59ff94cd285101d49b222c6a367fea6dfd651352b:action", "state_id": "599c2c63a629fb25997d884bcbee8ce13e161259aba92914a66347883be67a05", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.55615234375, -0.85546875, -4.00390625, 0.11328125], "student_probs": [0.26836466789245605, 0.19894540309906006, 0.008538564667105675, 0.5241513252258301], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "76a88cf09dade01f087e6900b4e8e2756895fe24d96539cc02c70769e2549230:action", "state_id": "cb3add5e2b828ac705dc18302fb75bb0526a7fc0e5b2de651a15e2194828dc05", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01953125, -6.16015625, 3.65234375, -0.978515625], "student_probs": [0.009178884327411652, 5.37334599357564e-05, 0.9812042117118835, 0.009563188999891281], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c558c3d18e50e779931e1d18648982d55bc5f6b0a52dfe07a7c4b6b4eafa8c05:action", "state_id": "c9ec450493cf0f1c8474b914931de0ad3f64e136f39cdefe6086679003687525", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.083984375, -5.5078125, 3.14453125, -0.583984375], "student_probs": [0.014029966667294502, 0.00016819476149976254, 0.9626704454421997, 0.02313150465488434], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "46738cdd012bf966ccd428e11f93754c90e5817017d2984af596a91b5a99ffb0:action", "state_id": "b76979c8bf2303c6abc3843cfe75c0cd25c3df6666734f3895d0afbdb41afeb9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.19140625, -1.9609375, 3.755859375, 0.55078125], "student_probs": [0.01816052570939064, 0.003094787010923028, 0.9405980706214905, 0.038146644830703735], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f90d6242afd437c11503cb469a8063d1bb0bbe453a6d11a5eb3b8b5dfcac6682:action", "state_id": "54be62912121d753e894fc59a4846e2e92d84084fe07dfb9d354979313e8d5df", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.765625, -2.26953125, 2.94921875, 0.6640625], "student_probs": [0.09233911335468292, 0.004438478033989668, 0.8198009729385376, 0.08342143893241882], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2e4f6f09bed2aeb32d8f1153f1c30ed01dbb62c7746fd0ce4695c7d0c52890b3:action", "state_id": "cf86ef0aaf3b278a46b8fb5224f43a68ccb857dc6a46133e3e62687c3fd5c531", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0703125, -0.673828125, -2.640625, 0.48828125], "student_probs": [0.32674500346183777, 0.1552504301071167, 0.021720198914408684, 0.49628427624702454], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8815824936110061c4e086043b8972aebf57c143a52a10e6e5ade45040303f69:action", "state_id": "c96adc58a068238a957f18d9155505d254c49df0ea3f5a1236e95f9b141f2061", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.380859375, -0.873046875, -2.51171875, 0.65234375], "student_probs": [0.2202606499195099, 0.13464263081550598, 0.02615269646048546, 0.618943989276886], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "13b4a61fb36d26f4940c6518670bb7f5627dbaf22cf4c18c228aed8efb774150:action", "state_id": "62efb6e26b19589b2784ba39ef9bad1bed516c72ffe4b2e2b73eebb835055cac", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7861328125, -0.765625, -3.921875, -0.0625], "student_probs": [0.24235977232456207, 0.24738134443759918, 0.010534768924117088, 0.49972406029701233], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "55b6189084333d996757369e192e722e627a57f55a5a52b4a4e0015ffbaeb179:action", "state_id": "44289837a29c70ce3301d1e03fb761f2d64bebd67d995067ea231ea5f7e7cb6b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.955078125, -4.51953125, -5.8515625, 0.1484375], "student_probs": [0.24688118696212769, 0.006989814806729555, 0.0018448958871886134, 0.7442840933799744], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "51af27bb9696be97fa0b5a33ffe8022eeaafcd10816ed94caa31ef6635578b30:action", "state_id": "230f76b4bc5fef4e38d04229101424c7ddbfb118bf105df38444dbdf49895687", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.609375, -6.4609375, 3.390625, -1.2734375], "student_probs": [0.00663041090592742, 5.182432141737081e-05, 0.9840402007102966, 0.009277612902224064], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "eba51db1c19f25f40d0d6caa6b8b49d5692b052f7fb9bcf03f0623edc515d594:action", "state_id": "10f71d8162195f4d5672d25115b57678fc27ec0d9d2ea9ef6f2ee7881ec98ce1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.140625, -5.953125, 2.78515625, -1.15625], "student_probs": [0.01898062974214554, 0.00015426536265295, 0.9621787071228027, 0.018686361610889435], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "aca4bc6813ce7e38d72d3d63bc750a1ef4a7499edb20c99d264ae2ad2e0849b0:action", "state_id": "e2a8783085c8f875100528418e735ffa6751050c2787632275559b97f43f7383", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3486328125, -1.001953125, -3.29296875, 0.41015625], "student_probs": [0.26964056491851807, 0.14029811322689056, 0.01419307105243206, 0.5758682489395142], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "61594c6b1168def4db4249d2e55a58d060c4ca37de1152f45797b15b637e00e8:action", "state_id": "508b8a1871da7cb51199bea09ab8f9633914b0638ae6b55f8e40ba10d8bb3568", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.140625, -6.1015625, -8.0078125, -3.89453125], "student_probs": [0.8368393182754517, 0.015937814489006996, 0.0023689446970820427, 0.14485391974449158], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "db1276243ea1fcb41a0e64ac477f196ce128d709f66c727385b9216c04db5944:action", "state_id": "0308dedc341b65bd26a1ff93e828779dde5eb75e8bce4aaae7d5783c2dd826d2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.41796875, -4.203125, -6.5625, -2.5234375], "student_probs": [0.9402354955673218, 0.009253821335732937, 0.000874294142704457, 0.04963638633489609], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c34ac172ca74023f8ae50aca0bb931cab2a46107cb393211a1ddc893f5e5b648:action", "state_id": "dd96d3f5ae8740c2a9eb9ba6e4a2a295beb9c23f938500f5985ef975954f1613", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.49658203125, -2.33984375, -4.109375, 0.078125], "student_probs": [0.33762261271476746, 0.05344574898481369, 0.009107842110097408, 0.5998237133026123], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "589b46baf35f2d6353c9ed1dbbaa4dc270bb79cf29a741dbe0b8072f552d9cee:action", "state_id": "73f9f7364e97a892483ea45f74657ac892cb4c9b60b300b7ca01921aa84008b7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.287109375, -4.765625, 2.80078125, -0.0703125], "student_probs": [0.04134929180145264, 0.0004693247319664806, 0.9068217277526855, 0.05135961249470711], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ebaa98ccd43f5dc0e288b543efb2638e0dcee979c82f939a6b6cdcff8820e48b:action", "state_id": "0d19819f65d68502602a3c3192b1c8c3ff1e255cf5e1448ed8a32e20457bddfd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.337890625, -2.98046875, 1.453125, -0.302734375], "student_probs": [0.12341874092817307, 0.0087846415117383, 0.7399618029594421, 0.12783484160900116], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "634aa1e62f31fecff6d839bd397e0f58996bed81f5b1352cf91bd85b5606c3a1:action", "state_id": "676a5a3a4d5a07425fba10e9f13c04b0a8577365c16f5b591028a13afe7dc5e6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.310546875, -0.998046875, -3.1328125, 0.3828125], "student_probs": [0.28068384528160095, 0.14113670587539673, 0.016692563891410828, 0.5614868402481079], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ad8107db8f4c18053005825df84e257d8bf6e7fa64089da7096013b6b0e88e81:action", "state_id": "93e55ae8e6fbb0edf96d63341e669760a8394e0eb4b3d3cefd27d9854e5c55c6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.955078125, -4.51953125, -5.8515625, 0.1484375], "student_probs": [0.24688118696212769, 0.006989814806729555, 0.0018448958871886134, 0.7442840933799744], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "96b88f5988dfb40f5d2535cef7b50bd877501ef5f890af7e2f07bf381bd1c1ab:action", "state_id": "8d1ed17c1cc2a7c3275f04bfefd7a5a480be4f820168638fe7d379ebbaddaf41", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.46484375, -6.4921875, 3.58203125, -1.328125], "student_probs": [0.006341608241200447, 4.157686635153368e-05, 0.9863461852073669, 0.0072706895880401134], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d4037dcd1c9f0602e17b04271a15dbc99836c7ba826b0992ae5bc171c43def8a:action", "state_id": "7ce0ad1098939db42550cad455ad2028c3043a7da145f4a6c07e0519ba7e6c3b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.36328125, -6.203125, 2.95703125, -1.359375], "student_probs": [0.012949350290000439, 0.00010240720439469442, 0.9739481806755066, 0.013000031933188438], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "546ff2619dcc58b7cb1e7fc19577fb05a9227d3e25bb1f85e7c2ec628701883b:action", "state_id": "cf182d4b39574b0998b93c13759ec8d8ee30f9ac022a48010f7ee12debfe6e42", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.1640625, -1.78125, 3.236328125, 0.265625], "student_probs": [0.03057071752846241, 0.006066944450139999, 0.9163819551467896, 0.04698038473725319], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e3ff64a08f61214978fd1d79da5a1354889074a1637cd891d3448e886e02d02c:action", "state_id": "fb5a31af85ba470bf5cb457296f1cc2d66b222da2212401521b4b1daaba51352", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.16796875, -0.9853515625, 1.62109375, 0.71875], "student_probs": [0.13648797571659088, 0.043073855340480804, 0.583685040473938, 0.23675309121608734], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ef524c9a317889b07d1c75ea35d084e43cc1ff8143201660776995580f5ee061:action", "state_id": "a84db2658c440730ed70820b9dbbc91a739c6375ba81f801611d9b40768ada95", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.283203125, -0.3623046875, -2.4453125, 0.46484375], "student_probs": [0.24085372686386108, 0.2225358486175537, 0.02771795354783535, 0.5088924169540405], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fc442a8f0a8f397e13ef2821bfb7a9ac770c0316d3818670e926fa19202a78ad:action", "state_id": "f497dfe2d3468d8281a76cbb0b5dba35efc46c2b3ede4a2e13c32cee4b16dd88", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.4609375, -5.625, -7.6875, -4.61328125], "student_probs": [0.6918712258338928, 0.07946664094924927, 0.01010304968804121, 0.21855902671813965], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ae979a0ebf981275da43420cfd04c4754d5144a199ff9672193e6abc71f78a18:action", "state_id": "a15d35509b280d2d0516eca7411789828959c683ac24e692e34623e0c87f59e9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.1484375, -4.12109375, -5.9765625, -3.5546875], "student_probs": [0.47543326020240784, 0.17975059151649475, 0.02810932882130146, 0.3167068362236023], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f12bc2efe3ba1b3cbbdb53505cde4ec4dc94e73d2730d52fc9e55a6db7945782:action", "state_id": "0a7efafd7db35f7cc3f7c41d905268478f40b6dc05542303cbc18ec8cbaafcba", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1484375, -2.94140625, -5.109375, -2.18359375], "student_probs": [0.40490350127220154, 0.18321861326694489, 0.020961999893188477, 0.3909159302711487], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d313f48c5b4a80aeaff34e423c6cd9250888ef4f833f00e310074383679bd8f3:action", "state_id": "cf5f331f5ed6e83f23668b30c3bd459162a07dab0ee1dfb70cdd89104a60d50e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.61328125, -2.375, -5.2421875, -1.61328125], "student_probs": [0.4010574519634247, 0.18723900616168976, 0.010646151378750801, 0.4010574519634247], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "670fc2c65fa2e1270bd5f9d1420038a037428d659021e4b7f446b19bffcc7b74:action", "state_id": "09712507bf0d8cea8ca30b679363c4dd6a576cde08037c289a66db6dcb06e709", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.30859375, -1.66796875, -3.9140625, -1.22265625], "student_probs": [0.7302097678184509, 0.10116666555404663, 0.01070462167263031, 0.15791894495487213], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4835822aaced97c53e41898c9819a845dccb841017594d4dedbbd91bb5aabc71:action", "state_id": "bc4bd757b467af858dadb2c5031f2ba1d82f00010323ea3b9147b1397090c3de", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4091796875, -0.95703125, -3.046875, 0.03125], "student_probs": [0.3121984004974365, 0.18051020801067352, 0.022330280393362045, 0.48496106266975403], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d5ca97a21b55a837bb9e78fc16fdb01ea8e8efb32908295dbbf56b7929c3e8ca:action", "state_id": "f18dfc4fe0c52af3bb80164a85ae9b96dfe5e7cbe581a4d7b394f781d17a7b20", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55078125, -4.86328125, -7.1484375, -0.369140625], "student_probs": [0.23256629705429077, 0.008471227250993252, 0.0008620164007879794, 0.7581004500389099], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c6077cd4137095ffe2de60fc3a47f246898ef9693c3a7faa91902408f3149698:action", "state_id": "3141f2745a7524b31f9f7eb58990cb6ddcb92cec0897ada7c9dcb8a0f5bc17be", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.17578125, -4.16796875, 3.28515625, 0.390625], "student_probs": [0.028879031538963318, 0.0005330864223651588, 0.9197052717208862, 0.05088265240192413], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b7c7493e7f7ecdd80327fec3cf5334b730de5446ac93a7d1e2a58d225505ff53:action", "state_id": "ba375229a7d30023aaf3d704202fe270b198047bd74ef4a5d412669bf381ab09", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.322265625, -2.96875, 3.0703125, 0.19921875], "student_probs": [0.0307711623609066, 0.002181676449254155, 0.9152122735977173, 0.0518348254263401], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "772255ccd3d0b4cd4e7d6bc26ff8ba114456cb32ba9361dac618396d3df62a39:action", "state_id": "e1ab65bed571b1827dc976499fd5376bc389efbecc380c9102af5b616aada696", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.296875, -2.1796875, 3.642578125, 0.4140625], "student_probs": [0.018322216346859932, 0.002787936944514513, 0.9415877461433411, 0.03730218484997749], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3a119b8636ed667ff0ea7f044d64af505789510cd54386784eff3ecf49bbda26:action", "state_id": "edd9e374d4dac745e80939cbcf22fa8b0c74e8542d07c9d5adf264b50c5dda16", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.6875, -2.328125, 3.0234375, 0.5703125], "student_probs": [0.08144927024841309, 0.0039922515861690044, 0.8421160578727722, 0.07244247943162918], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "286dbaf61a355d3216975d1c085cb09b4fef7d51b4cf0c10b7173ba0dca7d8eb:action", "state_id": "c038af54ed2fe66fdae53a9a8cae301406c181b3eb91f8eeb7ff53526c24aee4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.125, -0.66259765625, -2.4765625, 0.5859375], "student_probs": [0.32106298208236694, 0.14606322348117828, 0.023809265345335007, 0.5090644955635071], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bce002dbf64ca03bb0b7f588834e0e55565a627c0fb5227ae71e55ca8a5b85df:action", "state_id": "897016dc043dcf31666dc66660fa896c32f5cc126bb1d0ad5f8aa9264a1e44d4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.353515625, -0.43994140625, -2.73828125, 0.46484375], "student_probs": [0.2338583767414093, 0.21449574828147888, 0.021540826186537743, 0.5301049947738647], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5569348f27029cdd839292f205313f03918fb93b2a34387d92bb6a139c98ae49:action", "state_id": "0b26c5a5b372b2fb4c775183e729f1722a5e9a7a882edaefdd0eb9081238d662", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7158203125, -1.033203125, -3.3203125, 0.046875], "student_probs": [0.25341862440109253, 0.18450193107128143, 0.018737943843007088, 0.5433415174484253], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "94d0a35dd131b349d6c1edde91adfc2355fa28ff066da99c4ddc5f5d99f5c1d9:action", "state_id": "78b1d5e55c2e5ea2a68a4c3507cc4e3ef2023f47b178e7578970545d6375027a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01953125, -6.16015625, 3.65234375, -0.978515625], "student_probs": [0.009178884327411652, 5.37334599357564e-05, 0.9812042117118835, 0.009563188999891281], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "516c8a736a343c927e5c519e2d17be32c011462db0698cd79d77724ad692278b:action", "state_id": "2ecb981e2070f8ed7bc113ec45500d936f5f4689ac9b37b1b8eea5c2a9387153", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.083984375, -5.5078125, 3.14453125, -0.583984375], "student_probs": [0.014029966667294502, 0.00016819476149976254, 0.9626704454421997, 0.02313150465488434], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2061dd78f4c197fef6950e6738964df6c50d9725b2d2e1ce087c449c2eb3e53d:action", "state_id": "6623d1f4e280bbf28988add5850f3658e28a16a3e15c176bc6fda6270e03e740", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.19140625, -1.9609375, 3.755859375, 0.55078125], "student_probs": [0.01816052570939064, 0.003094787010923028, 0.9405980706214905, 0.038146644830703735], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c9528404111126270cd7ee0099ab454fc0e3f7f825d0b75168001d9c43df0c15:action", "state_id": "19706b2fd36f4b46faf7783d8a0fd15ff68411fd066040e6f6be14e75f1cea87", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.7578125, -2.26171875, 2.93359375, 0.671875], "student_probs": [0.09280277043581009, 0.004531011916697025, 0.8175055980682373, 0.08516061305999756], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "724ec5411c1c31a3edb3a64bf273a4d131b84548987451122fa2b29af754a2ae:action", "state_id": "efea3e7f8c5a1675b198030b28be2beec3870ee5fba8d5b7bd103dc99cb44677", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.078125, -0.67529296875, -2.640625, 0.48828125], "student_probs": [0.3285404443740845, 0.15466198325157166, 0.021669592708349228, 0.49512794613838196], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3d5ae6c69bb3576ad535b7d9f4026555f0f25e13eb9c671c968c27ab32ad3bf7:action", "state_id": "4de6f2aa0feebb83baa4b553d4526a6dbaa3b95f17ad12c84e423daf2a1bd8e3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.658203125, -0.84375, -3.05078125, 0.39453125], "student_probs": [0.20887644588947296, 0.17350319027900696, 0.01909000054001808, 0.5985303521156311], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f621531d8349efad06783b468c81675d709a2bd8b2f0c875f1b9976a09654a3e:action", "state_id": "28c8a695b2786c35a5c721cf0200464404b1479ed12165d8223d38d1658d7d47", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.919921875, -0.8193359375, -4.04296875, -0.16796875], "student_probs": [0.23413829505443573, 0.2589144706726074, 0.010307430289685726, 0.49663981795310974], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0e25ff9e68963b9fcf2f227be940d943f4d38acccac00fc7cb8c7708ba13f7b3:action", "state_id": "6edac89150e6b890e1d1a46f619411e8067c37dde50007be70905ef09c74c4a5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.19921875, -5.671875, -7.8984375, -4.40625], "student_probs": [0.7181087732315063, 0.06058000028133392, 0.006536502856761217, 0.2147747427225113], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ce5f9bd95419cadecf15c297827a11c0125571ffe5c87d131054d6c479bf2f0e:action", "state_id": "94697f494eae0e60178c858373e1202bab13d92aea03fb2e12491eff5d40936c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.859375, -4.17578125, -6.390625, -3.21875], "student_probs": [0.5011330842971802, 0.13435229659080505, 0.014667317271232605, 0.3498472571372986], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7139c69e1ed1572a13883fb0dd344a440896ac2e66901464f8814d38d0a3058c:action", "state_id": "2cc063a3b5fef0a68efbe30e3d879bcffa38169c9073c51f004454d63cf69d5d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.82421875, -2.6875, -4.7421875, -1.9375], "student_probs": [0.42216917872428894, 0.17806077003479004, 0.022815437987446785, 0.3769546449184418], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0e9912a72b2c11de18987e07ce7c1d450096041104fde17f26a1337ac3f05b5d:action", "state_id": "a0a661bb176671dccfb2858bd6be49a25a58080dd637ceab206c8a050618338c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.50390625, -1.703125, -3.8046875, -0.263671875], "student_probs": [0.6298756003379822, 0.06930319964885712, 0.008473372086882591, 0.29234781861305237], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c25977ac4e4dc108374a0bc18752aa7a343b7d5f9bd449b31edb2441f883c6c4:action", "state_id": "88d3b0c57d2178b4ca9168bb6e2d384e1b953b69e4a3e262279594ff21cc2d5e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.09765625, -0.98828125, -2.8515625, 0.38671875], "student_probs": [0.3669535219669342, 0.12387806177139282, 0.019221249967813492, 0.48994722962379456], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a11e5138563429c5002c8edb442c6c358fc3b9b70e9feb1e148dea48d14605de:action", "state_id": "52c6f2bc8b2dbf1c04ff8136b714a2f1ebf293708a42c55c0aea366158168810", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.60791015625, -1.359375, -1.19140625, 0.48828125], "student_probs": [0.19910936057567596, 0.0939149335026741, 0.11109194159507751, 0.5958837866783142], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5e45b621f276ddbd86f07560f4b28ca585d05a5e8a61175f396730a35526cbf5:action", "state_id": "6ee3b8571b177524f238320933f6e5776e7c9d1a15923c6b838674b3fa54a764", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.666015625, -1.7734375, -3.33984375, -0.130859375], "student_probs": [0.3218422532081604, 0.10633979737758636, 0.02220313251018524, 0.5496148467063904], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "db3284d4a099dc0d8ef963fb450d363640ad2308c10599c299a7cd961960badc:action", "state_id": "59cb86d756b3ca6372927d42ecd3b3282819ea0c142fb5a7a5d714c166f82237", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.853515625, -1.59765625, -3.6796875, -0.291015625], "student_probs": [0.3040034770965576, 0.14444494247436523, 0.018008921295404434, 0.5335426926612854], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1fde06d534d6b926612e0a10c09440786c282293fc2d617428e2514c92dc78f2:action", "state_id": "ddd0704ee3ed2df54d5c659a3c96f1a71666823b739cd0ad24e0b8cd640fcb10", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.59375, -6.0625, -8.0, -4.73046875], "student_probs": [0.7053371667861938, 0.05973546952009201, 0.008605709299445152, 0.22632165253162384], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6c0e5bf841fd4790887f214ca47b1ac17a95408313dba4782f64308a8872214a:action", "state_id": "b6cc482190dc9141f9d1ec072b334e2eb8ae8e95d17bcfc6ba3377bc6fa9e26f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.53125, -4.19921875, -6.46875, -3.171875], "student_probs": [0.576339066028595, 0.1087147444486618, 0.011236822232604027, 0.30370935797691345], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c6bc4d08c70efddec52fd3b4701a958281606095ac6db47a2442b66967489858:action", "state_id": "973fe0ad38bfd24ec0913a967975f0de7470bf9f5e1df8b4af984a24a5253b78", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.197265625, -2.31640625, -4.7109375, -1.7734375], "student_probs": [0.7474633455276489, 0.08979637920856476, 0.008190814405679703, 0.15454934537410736], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "09611be2cfd9d2869316551322d1346bf916ab16528318a1b21bfcd8c580c234:action", "state_id": "dfc1916579794eb320c7f3644323bcf2b455601bafe6235dbebffe42a5af3766", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.6796875, -1.75, -3.0546875, -0.06640625], "student_probs": [0.6304503083229065, 0.0555201955139637, 0.01506025716662407, 0.29896920919418335], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a86902ea9dfc4d8e9da97b805ef61241772d119ef3b7ff6a9ec8b82634113598:action", "state_id": "4443a03b799d5d341acd747eb83a6ac03fe1ec757f738fe00c6282376c66b355", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.234375, -1.34375, -3.671875, -0.4541015625], "student_probs": [0.5784297585487366, 0.11936572939157486, 0.011635574512183666, 0.29056885838508606], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "67d7f260a73a9033b0168a3b6dd71a6e0f0f69369993c6c003e28fbb92be4191:action", "state_id": "09f85b7a9d1ac6d7dcd39d078480d9f1207ac6fe640bb6250791865089429347", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.33203125, -1.26953125, -3.3359375, -0.046875], "student_probs": [0.36085861921310425, 0.14131426811218262, 0.017896050587296486, 0.4799310863018036], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "035b1a8e56a54c6ede7b60c96e0aa69d0a079fff8d80b19b2bd98164d5fb3856:action", "state_id": "7b6b7000f009c309c10e3f99a0c2839716501d53798de67f3830ec00a35ab28b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8134765625, -1.4765625, -3.8515625, 0.16796875], "student_probs": [0.2363230586051941, 0.1217675507068634, 0.011326146312057972, 0.6305832266807556], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "487560553063426d13089e5d50a839557011f43d8e7f9dfda9a883f8c8e947f9:action", "state_id": "7133f0a127ad9448850bc46b9345518598262375b5fc4962ad86a44dde95f49c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.1328125, -5.6015625, -7.8359375, -4.25390625], "student_probs": [0.7043837308883667, 0.05965472385287285, 0.0063865757547318935, 0.22957493364810944], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "296573d49cf37ccb7cb0df563264eb31c53cab53b39bcbc253fca827c6aba1aa:action", "state_id": "6c7d2319a7db03ae576c2dcf62f2f14c7bac8c2ae33ea493d4dd5cfc26d3b048", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.7578125, -4.078125, -6.3671875, -3.09765625], "student_probs": [0.49850401282310486, 0.1331264227628708, 0.013493885286152363, 0.3548758029937744], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bd63c482850a6e05856883b37e962a02aab5c82475f0731b2f3153a88e933df9:action", "state_id": "dc49806caf435679a250e92b782960a1fea8692ba3eb9432bdb903b510f52002", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8671875, -2.7578125, -4.8984375, -2.0546875], "student_probs": [0.4371234178543091, 0.17939509451389313, 0.02109351195394993, 0.362388014793396], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "84bcbc6c8df783b43c14db8361628dc911d20c3b41d83ee960836c91ce7f1068:action", "state_id": "a66ba638ad69f47c9bd14667119a66ced4f4101c4e1012e509de777895042bf4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.203125, -1.80859375, -3.93359375, -0.4990234375], "student_probs": [0.6078091263771057, 0.08129968494176865, 0.009709862992167473, 0.3011813163757324], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "757b275c76eccaa832f93a3276cc38618852d93af7d90a44836e63d4d9eed267:action", "state_id": "6101d8decd38bdd5f4c32638a8ec1bb0903faeab6814dd28fe4e640f834d8d3a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0546875, -1.03515625, -2.984375, 0.359375], "student_probs": [0.3649168908596039, 0.12271025031805038, 0.017472131177783012, 0.4949006736278534], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c4d11cf46018293f560e1b9ea21ed4fa7e23449235bf3e1f6b54c0f8cbf4d56e:action", "state_id": "7ac942c5623bd6abaa53ce0201c85770d9171ea3e38577147af2190947ae6ff5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5029296875, -1.0078125, -2.86328125, 0.2734375], "student_probs": [0.2582961618900299, 0.15590143203735352, 0.024379808455705643, 0.5614226460456848], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0a139e6fd68ff304987aacbd54698891d5dbcde1499e4d9d1a9ca19109c61000:action", "state_id": "ab79340486530187800507163862925d7869ea5f9b0cf78b54aa961a8e2417d3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4423828125, -1.291015625, 1.33984375, 0.0078125], "student_probs": [0.11186067014932632, 0.04787633195519447, 0.6647962927818298, 0.17546671628952026], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6793b9f39f63750568a1c096fc7d8b76b2212c462b8372c86069c2d99de8d4cb:action", "state_id": "5b117a9a60a3ae9aef3bbdae614e2c7b5feb3681b72b4c1d3d10e1e4e725464b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8251953125, -1.80859375, -4.01953125, -0.224609375], "student_probs": [0.3088117837905884, 0.11550728231668472, 0.012659350410103798, 0.5630215406417847], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "336c6ca1081cee5ab80e34644f7dbd8b10ad2cc64752d3487437392ccd3d242e:action", "state_id": "e101fd397d30e0ead0c053c3a96503202ec2e0d9afa3a41a00a4e60c26eb06c6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.916015625, -6.11328125, 3.65234375, -0.8857421875], "student_probs": [0.01016031764447689, 5.620351657853462e-05, 0.9793108105659485, 0.010472608730196953], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "374a6f23ea70a915cbd98f8f95de5ac3f24abe1a450126e8802ed5f106d764ff:action", "state_id": "46ebcde3281519b172373e955a14961f74ccddf1532007f4d9efc09554954f97", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.29296875, -5.890625, 3.16796875, -0.96484375], "student_probs": [0.011240114457905293, 0.00011324889055686072, 0.9730412364006042, 0.015605352818965912], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ea15080a585d6ebb86d2be14910c364c1a89437d8bbcebeb8eb4c588714bdf63:action", "state_id": "ecb7a5fc5715eb624add1c88473b5d8cc7a1315d60ac5e72fad1cd125faf6781", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.189453125, -2.01171875, 3.78125, 0.5078125], "student_probs": [0.017796216532588005, 0.002876920159906149, 0.9435874819755554, 0.03573932126164436], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "22c2a9887fcf33dc398154439ea485d369fd0c0e6b1b48ab41d8d1f1e62a54e1:action", "state_id": "acbed2f5a7893da673825cb7b001286aef2a1d0c28164d82881d94ac85221927", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.67578125, -2.2734375, 2.8359375, 0.54296875], "student_probs": [0.0943351536989212, 0.0049413335509598255, 0.8181209564208984, 0.08260262757539749], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0fc8e55657d4e1f053c7c1b6a9248044c0f4a687e5b3eda3e28610af89bc6b10:action", "state_id": "dbedf74de2649fa58034344546b0b56f3286302e3e0cd7a48ff13057f4b51c60", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.03515625, -0.6611328125, -2.6953125, 0.52734375], "student_probs": [0.3125477433204651, 0.15578365325927734, 0.020374590530991554, 0.5112940073013306], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9eb5c2fead215535e0636349aa530966d9cebb6803728340c3f008003083cbf8:action", "state_id": "3f7f793af362cbc672bd7f88814b7bb02f3cefe4d09c5f6a627f67b048dc9bd5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.296875, -0.92578125, -2.54296875, 0.2734375], "student_probs": [0.2934439182281494, 0.1564568728208542, 0.031049814075231552, 0.5190494060516357], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1781597294c5e90f1c485550972d5cc9f149af8cfc19b4eafaa96827974b59a1:action", "state_id": "0a8fab352b616182ce3017dd9392814b5aebda0add59085704a863b065ec054f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5867156982421875, -0.6728515625, -3.671875, 0.13671875], "student_probs": [0.2484661340713501, 0.22796010971069336, 0.011360554955899715, 0.5122132301330566], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "12c386f09af33aac69f14a0cfdc335dc0f0c4813d1153423e38f927c1905d00d:action", "state_id": "7869eeed71e0422356aa68b58397fcfa0656a48393f9dafb516a9c8ce54fe813", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0390625, -5.01953125, -6.8984375, -1.1796875], "student_probs": [0.767504096031189, 0.0048770965076982975, 0.0007450110861100256, 0.22687377035617828], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9b9ee3e6060b98f07496d7529651c3cc9963e1ef764901bcc5a6dadaaa5435bd:action", "state_id": "90ca034a15628fbbbd09a3a4e5cbe6d57f9c8bb51309ddf6bde9d3f2b555e75b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8447265625, -3.7265625, -5.5546875, -0.54833984375], "student_probs": [0.41493308544158936, 0.023249445483088493, 0.003736526006832719, 0.5580809116363525], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "360711f0ff4f910409f2e961bc2265ce4ff76c987d8d033a9da3e7446900ac0b:action", "state_id": "f73d09f85cb28d7aca993decf39847f60983304207c9241dd6d61fc14b104396", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.451171875, -1.9453125, -3.6171875, 0.2734375], "student_probs": [0.300251841545105, 0.06738894432783127, 0.01266200840473175, 0.6196972131729126], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b6b301f24ad1ffe101424950f1039ac618f1195ec20f9ab3383fb485c03ee89c:action", "state_id": "abfdc9059687fc55be9ac891d590eaf05b093a6636d74e09cbba01dbdef4c1e9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.46142578125, -1.3984375, -2.79296875, -0.0078125], "student_probs": [0.3264846205711365, 0.1279156506061554, 0.031716588884592056, 0.5138832330703735], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7e5198d55066f53cd1957860f0d16cc5b46de21e07dc5802a80aab49504b7ad5:action", "state_id": "fcddd1df0ee048e96b268be4337c563238c4016c8910c4af65fb8f5a35d0e0df", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.02734375, -0.900390625, -2.61328125, 0.9296875], "student_probs": [0.25431445240974426, 0.10056829452514648, 0.01813686266541481, 0.6269803643226624], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "90927331d01a25fc35443671a769618cc7316650697b11999872a93df344f76a:action", "state_id": "6a5c301ed7da605d0d3354b07d994f47fda2feaa226e9538daeb1b05125e1b26", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.421875, -1.41796875, -3.24609375, -0.0703125], "student_probs": [0.35088199377059937, 0.12958748638629913, 0.02082660421729088, 0.49870386719703674], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "077a9c1b9b6704fb6c88bd079a34d02749800de6c66b29711d4f5f8f3a3b60fa:action", "state_id": "124d30f4111ece71ed833bbb02a876b3e1bb2bd19b75d1c502cfa377b587fe6e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.005859375, -4.53125, -6.25, 0.23046875], "student_probs": [0.2233305722475052, 0.006574920378625393, 0.0011788182891905308, 0.768915593624115], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "54692fe4e2f8b737efe620ca38ce709354126f8528b649db873f82e17c9d241e:action", "state_id": "283016c21e5de2f89565e9147c646601cf95ab501c82d4cf864037d3408f73dd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.53662109375, -4.4375, 2.7421875, 0.26953125], "student_probs": [0.03355296328663826, 0.0006785794394090772, 0.8906341791152954, 0.07513432949781418], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d3cc7dc084c52bf5b41110cb2825cdf6a494be7a96eb7bec226b8f120f741e91:action", "state_id": "8943ae2b1e0273b38c0945d99047a0bfa6718e52ddf63c704b7f5a54bb937e2f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3955078125, -3.26171875, 2.984375, 0.1953125], "student_probs": [0.03102727048099041, 0.001765891327522695, 0.9111880660057068, 0.05601876974105835], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ec482b259aa19685b4aa781bf95327564d4db34ce1d3eb317f9d9f53d1c583ca:action", "state_id": "db20e6b4c272e8b4b54dec021ca831556b7ad9119a06cab6aa5e9820dcb1bef1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4150390625, -2.5546875, 3.5234375, 0.3203125], "student_probs": [0.018333742395043373, 0.0021578120067715645, 0.9412603378295898, 0.03824813291430473], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2e1f1711b4ec36eb99522627714bd8838a54296301ed952d4998e453163f37ec:action", "state_id": "1d09e20c035f1f913e339a8aa2b7cfcf809463a3577f20d2749179164a7af7a8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.7109375, -2.25, 3.080078125, 0.65234375], "student_probs": [0.0788453072309494, 0.0040818494744598866, 0.8427146077156067, 0.07435820251703262], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "151c18212ecf621945c0f3b1361ef56c1d580729e1affec767766fba4eaab36e:action", "state_id": "53cfb3cb99bcb14c88889c3177c7a84f787246cb55c1ae00faef994a64f05469", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.05078125, -0.6845703125, -2.8046875, 0.47265625], "student_probs": [0.30468523502349854, 0.1616591513156891, 0.01940193772315979, 0.5142536759376526], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ff03d0aba051215be256c080004afd0ad27c2146b2adc257b0b0db1bddb37c90:action", "state_id": "6e597b788691ca305cd331fbb955c5f99a6a17c52ca03d8b449a4b1f2c89fe51", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7255859375, -0.4296875, -3.12890625, 0.01953125], "student_probs": [0.2201945185661316, 0.2960149049758911, 0.019909381866455078, 0.4638812243938446], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "771eb2bf7735814dec9ecc01a5404f7f0730d9e18780ac1db36757269ee12d17:action", "state_id": "1c629d72e2db4b76e3947aa7584d679e1c528bd3942d365327fb1806017eef67", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.95703125, -1.1484375, -3.70703125, -0.162109375], "student_probs": [0.24366514384746552, 0.20121794939041138, 0.015576992183923721, 0.5395399332046509], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3eb124c2a6ae0382b1f17cdc307b26b52c8025084bc4f5367b46c026486306ae:action", "state_id": "ce1f06c39179c038a5b4fb378eb88da748a07b3a127bd17e35e0f587783eb520", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.109375, -6.140625, -8.0390625, -3.890625], "student_probs": [0.8411568403244019, 0.014932322315871716, 0.00223689922131598, 0.14167392253875732], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f1c21123061cb337dab47b055087a93a4b4a383d792b57f2e57a31672607ee61:action", "state_id": "e4c3dd2262179900c0728c437c5fa81741089782f3224f819f12884fc0c9a8f2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.69140625, -4.359375, -6.6953125, -2.5625], "student_probs": [0.9563458561897278, 0.006124752573668957, 0.00059238460380584, 0.036936987191438675], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "60976988c65f0a77d7e1cde4c5e995ce2bea1617d62edc1dfbee7eecddf5bc65:action", "state_id": "f551d0c6ae4e04d0a6e33d56ffb238a2b164650a8192ab8fbf9688fd0c3568b5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.51953125, -2.33984375, -4.22265625, 0.04296875], "student_probs": [0.3399422764778137, 0.055062185972929, 0.008378347381949425, 0.5966172218322754], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5c179023b6b2c2da91a8146bf01464e6a022f804ac2402879dc5517168af6b23:action", "state_id": "cff35aaf568febf4fc3efdc2174a38c6ef0fc3786245226d95409f4cb85dae9f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.96875, -4.53125, -6.2578125, 0.2265625], "student_probs": [0.23052074015140533, 0.006539369933307171, 0.0011633203830569983, 0.7617765665054321], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ec5d7401d657bc5345eea178d7dc8adbf4a8112969a060d27effe37b8ca5e092:action", "state_id": "0d8b7e0b1dd5341d81a3b02ba6fe9060daa80fc48851c33c44ccb46db524bb5f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58203125, -6.578125, 3.484375, -1.35546875], "student_probs": [0.006216417066752911, 4.2049825424328446e-05, 0.985944390296936, 0.007797133643180132], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c893326d230dc6f6a8915cdb1468939fb30d13aeb85e04064007473431ca8d71:action", "state_id": "fbac0b59e984ba81cef4bec75d7dabc44ae12569e4faf9069d91810a17bf7eb6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.556640625, -6.2734375, 2.96875, -1.498046875], "student_probs": [0.010593078099191189, 9.474216494709253e-05, 0.9780799150466919, 0.01123231090605259], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b811dc1b6c631a3c87eef179255106d57d696cfdf37d0ae0e831526cd1a0b6b7:action", "state_id": "73df7b40d01bc5cbfdf493cbc4e1b6d8591be41291b15a8b672b3a73950adb4c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.08984375, -2.171875, 2.970703125, 0.33984375], "student_probs": [0.04166548326611519, 0.005194715689867735, 0.8891091346740723, 0.06403056532144547], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b646133c0faaa1febccc8f18b0a237796446a0c37eb07997f10193f777b7e622:action", "state_id": "879fc6801c3e217c07fb55b8742b7440e29f2928b823a15465cd5efb7226dbcf", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.41015625, -5.8515625, -7.984375, -4.56640625], "student_probs": [0.7082069516181946, 0.06164117902517319, 0.007304697763174772, 0.22284719347953796], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a25377bff4d5ededb38b4cce5401b31436eae65de8be546dfb5d35557c09ef6e:action", "state_id": "af94765846c294f4a5846bfaa94a366b5a7ccf4581535672b9a62e79039cb774", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.90625, -4.23828125, -6.390625, -3.25], "student_probs": [0.49907177686691284, 0.13172529637813568, 0.015307989902794361, 0.3538948893547058], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a3ad7a8ec72bd569a71b9f567a27754fc6160e6ce1fe59a22fa689d978e4ddc0:action", "state_id": "40a5c9912442d188c18ab1f7d826bce78240e59d1e5c343d5b52b76e801b828d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.70703125, -2.63671875, -4.7890625, -1.96484375], "student_probs": [0.45181748270988464, 0.1783219873905182, 0.02072305977344513, 0.34913748502731323], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b82bc0046e73267b4a0de3d76d4743d75a743f9b15c2d8ce2a32eaa2cb2ed6eb:action", "state_id": "e3813c9899518565db6ef931e14c2edf56fff55ef01fed1ef70a84673581d007", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.09765625, -1.875, -3.9140625, -0.61572265625], "student_probs": [0.6071000099182129, 0.08443966507911682, 0.010989879257977009, 0.2974703907966614], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9dc9f6e4196d816632be1b8420af2759d8b7ddfd7929e5f81a4c94044512384e:action", "state_id": "24c1c852164dc47fc255caa81df34d9dac2144c71217ebd43c49494d4ceaf192", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.228515625, -1.25, -3.28125, 0.16015625], "student_probs": [0.3469439148902893, 0.12492065876722336, 0.01638602465391159, 0.5117493867874146], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bab40a13bc4e0a1403210671df8c36f2949e22b5f8d0345bc5ce1d6c5654b0b0:action", "state_id": "08905ceac63cb8f3fa31dd1e3ff44e8799d5f363438a31e51ae9e16789a97521", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.271484375, -0.6181640625, -2.765625, 0.31640625], "student_probs": [0.2785681188106537, 0.19695650041103363, 0.02300063893198967, 0.5014747381210327], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "531962c870f64d8a31b6fe4d12636cb891566fee0931bc131cbedbfba1df727a:action", "state_id": "dc07374f736788ce25b4edeb668a89638e2d8a2739bedd46e92e5485f19b5031", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.67333984375, -1.30859375, -2.28515625, 0.23828125], "student_probs": [0.23709721863269806, 0.12561433017253876, 0.047306790947914124, 0.5899816155433655], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1970b0cdc10af94e64981a55272bcbf8ab6e26cbf1d1de36bd0286b21a377579:action", "state_id": "ff13cd442883e9b9f0dae5e2a26812ae0293bc42f13a01bbe564a54d7d067250", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7177734375, -1.79296875, -3.23046875, -0.1640625], "student_probs": [0.3162578046321869, 0.10791698843240738, 0.025632532313466072, 0.5501927137374878], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2fc8650b2155885023ff4ff5c20d39ef0dc03161f45372457bd24d737833c620:action", "state_id": "c6c163ff7f4dec22f2ce4d5e02d35c972bc2d687b3642dc97f73d7afb660dcef", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.98046875, -5.8984375, -8.0546875, -4.34375], "student_probs": [0.7598096132278442, 0.04106266051530838, 0.00475334795191884, 0.1943744719028473], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6e3eb052884db5bb606b5628e915d5adf441c844145a467a863cc6b08efaed52:action", "state_id": "f078f2aa999b7e60f84efb9dd58d399f6f4298034dea5cb1d7f0708589896815", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.48828125, -4.45703125, -6.5625, -3.4140625], "student_probs": [0.643973708152771, 0.08991887420415878, 0.010951092466711998, 0.25515639781951904], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ea1629dd458a68f96c3514fdb17d1a0f963c2bc34cc7a3676cccdeb84835c123:action", "state_id": "d671c420154bf54c80c5139b455524b31166fb6e1b6970d3bf547d48bd2223ea", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.24609375, -2.53125, -4.5390625, -0.6124267578125], "student_probs": [0.6691895127296448, 0.041625939309597015, 0.005589618813246489, 0.28359490633010864], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1ebe8f08a69bedbe3df04b4fa54c534968540a71026e61360839ab560d205c05:action", "state_id": "96ffd9bd6fc4810f3b4e303d02a8619f47d88443a15663e47827e4dad69f7276", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.09375, -1.70703125, -3.34375, 0.51171875], "student_probs": [0.3257203996181488, 0.0648941919207573, 0.012629549950361252, 0.596755862236023], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "933d3707bdab16add292622055d8e1bc3d2f88d41f78714cbb11fb3424e4f3b4:action", "state_id": "842230586c1eea0522d20c74a75125400f6d0a463f83eb4c07fdce19161a55aa", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.66015625, -6.125, -8.0625, -4.83984375], "student_probs": [0.7119234204292297, 0.06052924320101738, 0.008720063604414463, 0.21882730722427368], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "431bfa3abeb3e7dcbb57089b7debea3742c0b80ddcd050491edce9cbe7d30724:action", "state_id": "79a1d1934142fee84c64331afdff1461062a35741997b779d15689274a84ba1f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.61328125, -4.33984375, -6.6015625, -3.23828125], "student_probs": [0.5774713158607483, 0.1027291864156723, 0.010701431892812252, 0.30909812450408936], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ebe8f62a2892f8a986124e360117e0e8ed7aced3330cf8811f4ba053b3209088:action", "state_id": "9db042370b474321845dd49b1948f8f2b72ae4dc8b954ed5bc2519b38816506c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.03515625, -2.359375, -4.8125, -1.6484375], "student_probs": [0.7659835815429688, 0.07495905458927155, 0.00644830334931612, 0.15260906517505646], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "58571e93d6a5b321a0bbbc18173e1bc6c2b4fcba48f200578a72954e3767d66e:action", "state_id": "86fe80effb611f5512e6ca903a0ab9107a9a9355d3115c6ccd613617608021af", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.263671875, -1.640625, -3.39453125, 0.015625], "student_probs": [0.3819379508495331, 0.09638060629367828, 0.016683142632246017, 0.5049982666969299], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dba97da80826f74f71b3416d2f765a5ffd258d17977ef86f996b96282ceecc82:action", "state_id": "5a4952af801c74cd3ec616e632cf202e7042321b2d3776df07711ed98bb77729", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1875, -6.09375, 3.62109375, -1.1328125], "student_probs": [0.008024215698242188, 5.938070171396248e-05, 0.9834411144256592, 0.008475261740386486], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e18e91001fce69bb2d625994932a83ea63bae5d9fc30f2260d86e428f0e3c280:action", "state_id": "1d74d42ed7de33dc886c476bfe93e9b88cacb2aa7c82b27470d1de2df205aa5c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.19921875, -5.53125, 3.125, -1.060546875], "student_probs": [0.012875251471996307, 0.00016919146582949907, 0.9721651673316956, 0.014790407381951809], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9ddec3497a75df9a076a4334f860235833b2d75d3aa0c05da7d82685fe672590:action", "state_id": "0880e08739b69f5d692eb557783a0da48533bd3908266d914970c7d14317e173", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.181640625, -1.953125, 3.28125, 0.21484375], "student_probs": [0.02893036976456642, 0.004920486826449633, 0.9231414794921875, 0.04300757125020027], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4dcb78e6adb2105bbc70d2a4e38ad01343db8288673bf84ef86d25bd2947c078:action", "state_id": "83ca9eed362320980e6211d10c51cb2201eb7527df9a056fe1324d6f7b7b9a4e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.20703125, -0.9814453125, 1.703125, 0.73828125], "student_probs": [0.13386954367160797, 0.04078805074095726, 0.5976226329803467, 0.2277197539806366], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b12d71c8ccbf42da3e385ab91eb60784d8de5d89e20b0c7b56da853586a7232a:action", "state_id": "b80de5e80b4622cf32117d7497b508ed2a4c6d6db97f688177368e3d433b3cb1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.392578125, -0.3369140625, -2.546875, 0.4140625], "student_probs": [0.22657260298728943, 0.2395421713590622, 0.026278959587216377, 0.5076062679290771], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4e3c630f6dce7a21dcea4b5a0c80e407019b3e11d36f083d51ab2a68ab45684d:action", "state_id": "6de25cabc8626baa001842700dafde9aaf81c497f1e98640703b8d0fbfa4a077", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.005859375, -4.53125, -6.25, 0.23046875], "student_probs": [0.2233305722475052, 0.006574920378625393, 0.0011788182891905308, 0.768915593624115], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "eade409fe6c2dafa6c6c718afe8cb8d8612da7925c2cf570f488cafa162539a8:action", "state_id": "a50888c66f31a69e1d91d4030a1323969ea4e43ee252468fa5ae101aa5ea1b1d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6552581787109375, -3.20703125, -5.0625, 0.3203125], "student_probs": [0.2671787440776825, 0.02082480490207672, 0.0032565752044320107, 0.7087398767471313], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d91650251418a3b6c442141c6e32c7e15f95fcc9d17b6de717fdc566f5604440:action", "state_id": "31d017ae9301f37537db6030a8ccf8e7daf695acbc0e9c070b0ec6f6482a482f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4521484375, -2.38671875, -3.22265625, 0.09375], "student_probs": [0.34091585874557495, 0.049257684499025345, 0.02135162614285946, 0.5884748101234436], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f09d3d47c39e28a8b90449e968680b035937299ee60570a67343ac8ba944db04:action", "state_id": "ff320ed24ab87f317cf2b5e063b83d10e073e72ccce4711bdc059f2da448bdf2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.359375, -1.63671875, 1.48828125, 0.79296875], "student_probs": [0.17328232526779175, 0.023542998358607292, 0.5358361601829529, 0.2673385739326477], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2811e409b84d7dc375c22afe6d34221df56f63c86673ab94e78f0ba6c8a26388:action", "state_id": "8b72e522d6faf9d52a34934659f6e1f5361599f6ba17c865e10817ea0fa83bd6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.01953125, -1.390625, 1.333984375, 0.22265625], "student_probs": [0.15627752244472504, 0.039667800068855286, 0.6049519777297974, 0.1991027444601059], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "225d1866f8eb4e20aad2b9a3b9d210d3acd50c3bf3446435de2acb2837576339:action", "state_id": "c9c0bc92dcc833a676a1302951b3c518c16d5c690ea654b88f2514667ec0bbd1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.24609375, -0.99609375, -2.82421875, 0.37109375], "student_probs": [0.2939456105232239, 0.13885007798671722, 0.022315237671136856, 0.5448890924453735], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8c45f553fecb451522a48f8ab313eaaa4bff47705de121fda06c4f092502382e:action", "state_id": "f69c942e68fc966f13df7a14e7f7b6f972ebc1bd2ee48958a920c747b91e94de", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.724609375, -1.8359375, -3.84375, -0.380859375], "student_probs": [0.3592544496059418, 0.11823837459087372, 0.015877297148108482, 0.5066299438476562], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2920fce56acc0b553962f1e631138e2a48e77c18dcb8c3116974b2596a0bc41d:action", "state_id": "58e430c20db336edda3f44885b19b24efc15decc28ae4d0df5663dae018d2021", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.6796875, -1.1171875, -4.4296875, -0.68359375], "student_probs": [0.700438916683197, 0.11614415049552917, 0.004230550955981016, 0.17918626964092255], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "78da56b30882df0315319e38347f98d64c8d2de0b6165c62bb83494f28c418e8:action", "state_id": "a9cf4811ff4be03255070fc90a2e733b1b212bb5f494f9239b6e07252b369373", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4814453125, -0.93359375, -4.3671875, -0.628173828125], "student_probs": [0.39677485823631287, 0.2524518668651581, 0.00814681313931942, 0.34262633323669434], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6d961df46d1ab14c0eaede34e318a1e3f70ae7c4b60485abafdd59ddba22e560:action", "state_id": "77a648bd9ebec803264216914f3aaa59dcf1fe11e818d81c3ae38a607694fb66", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.421875, -1.875, -3.94921875, 0.46875], "student_probs": [0.27027952671051025, 0.06320172548294067, 0.007941585034132004, 0.6585771441459656], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0e3ae0942b6c07b8c632488735202df1423f950a102aa953a1aeca99c40be758:action", "state_id": "3c7415bbb22d04f1d55bbce66e7f6951a3b204b4ae4074d804fa3d9db98ad793", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.65625, -2.65625, -3.78125, 0.0625], "student_probs": [0.3094883859157562, 0.04188470169901848, 0.013597970828413963, 0.6350289583206177], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6af01b62f338f3a0d2b75531fc149ecfd231a436a32aac907affbe3617833729:action", "state_id": "f9719816274da86716fa77c82db3befd21f2b3276103844636a8dab3857bb498", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.142578125, -2.40234375, 1.87109375, 0.03515625], "student_probs": [0.1021486446261406, 0.010661759413778782, 0.7651722431182861, 0.12201738357543945], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4b8bf66fcb1a177d69e8190217877253df656348c8ecf033f08da573cb17e721:action", "state_id": "2e789147f624cebff7eacd85d743ef5d1e1c68ba3d92c103625f5a6947de79fa", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.14453125, -2.03125, 1.64453125, -0.095703125], "student_probs": [0.12216801196336746, 0.018516801297664642, 0.7310338616371155, 0.12828128039836884], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6326fb4cb42fb0c618b2ec5982566be636889dde65ce458b0bf8cbacaf6c05e4:action", "state_id": "492171af42aa17dd0a6db371fcf556f4f16b20c4dffd48503b2679845811f6c1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3203125, -2.1328125, -3.72265625, -0.046875], "student_probs": [0.3982452154159546, 0.06501173973083496, 0.013259630650281906, 0.5234834551811218], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a72f726c1ddfa01ff7f31ca1c2c24d4f5af5f90cfbd2c14700eeeb4a9b49ec37:action", "state_id": "7435cf2c3b664e0118644e5777ceed96367014c57b2b30a73181cff3f7c26982", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.28515625, -6.05859375, 3.71484375, -1.29296875], "student_probs": [0.006648324429988861, 5.6186847359640524e-05, 0.9866988658905029, 0.006596587132662535], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c87d519374384ee28199f0f8bc6921349cd5fa15bda13f2f50e87972db0be449:action", "state_id": "2e4dca3a09589a4d5c6d0bc2337882a73c55c499aa80fff09a5c3d1c8ed30508", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0546875, -5.0390625, 3.2109375, -0.8447265625], "student_probs": [0.013612545095384121, 0.0002532487269490957, 0.9693413376808167, 0.016792843118309975], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "20aaa23718a74d1865fba8385387a6c2d6c4acdb1f921be43a680dea46ce4e1a:action", "state_id": "837f239fe73cd745ed70fed5ebd5895950be758c7c0d7c66daaf6f801fd6bc52", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.1484375, -2.33984375, 3.27734375, 0.28125], "student_probs": [0.02994443103671074, 0.003346573794260621, 0.9206910729408264, 0.04601791501045227], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "171dc61cb9888bea9adf52a31b3f2f2bef03f5d820f3a2f46ec56f77c7ff44b8:action", "state_id": "f4d18dae4aca87c7baaf323250c666b78dabaaa8b9ff6efaad66a0c54341dcca", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.09765625, -0.497802734375, -2.734375, 1.0703125], "student_probs": [0.2350085973739624, 0.12956246733665466, 0.0138403857126832, 0.6215885281562805], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c057cea97f06ab18c2ee221bd531e2c9b547e828f6269a32a28d82d2062313ab:action", "state_id": "1f56ef1066f832b894d114ab5c046a34d93923d5d5053d23af9946285bc56ae2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.294921875, -4.71484375, 2.69140625, -0.109375], "student_probs": [0.04539530351758003, 0.0005463403067551553, 0.8994080424308777, 0.05465034767985344], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9b7d063b71f438ce667920a782531221b25326b8f4bf4b624f422cb742da12d7:action", "state_id": "bf1de44ed8e4c4c80927325274a657bce3ad8e8ee7e79b8fae6ae8a54080eb59", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.38671875, -2.96875, 1.64453125, -0.3583984375], "student_probs": [0.1027965247631073, 0.007773498073220253, 0.7836806178092957, 0.10574936866760254], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c427a30fbd5ac59304ad6f772c1186605f6223c1ccc10d1ed4a9237b32b739ec:action", "state_id": "ad33d4e2c5fc2600b426b19ae86652d5b5a376e28c6255ef339c82c4a90a8f59", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.52099609375, -0.982421875, -3.44140625, 0.3125], "student_probs": [0.25090035796165466, 0.15816360712051392, 0.01352643221616745, 0.5774096250534058], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "af44110aab6977b2f93f5763cd35dcaafeee3d7e86e1bee2898fa67a44d51874:action", "state_id": "a42bf85b7a22192112dd532969867552d75dea6e65ad5fe4b349f089eaee64a5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.59307861328125, -0.9765625, -3.125, -0.283203125], "student_probs": [0.32007738947868347, 0.2181273251771927, 0.025448109954595566, 0.43634724617004395], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1f7b0ed471fbceea0fe486b7fc36f68be8abd5ea1528c55cbf879d59119b89f6:action", "state_id": "8825815806f582e68a120a4fae4eeb0bac35da487669dd0aa5756aa16593fa97", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4384765625, -0.50830078125, -2.91796875, 0.51171875], "student_probs": [0.21727047860622406, 0.20261727273464203, 0.01820417307317257, 0.5619081258773804], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7c887125687bd97b440e67cccea53f4d4f5defd861aec40b8623f8c546dd1233:action", "state_id": "ce19ad3ca8ffe55bac5d33d493e63408bad06354172e599f5a25c1d66f49a3e9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.978515625, -4.51953125, -6.2109375, 0.18359375], "student_probs": [0.2363511025905609, 0.006850370671600103, 0.0012622508220374584, 0.7555362582206726], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3d80a929f9cfc1aec0cf0f87768b9b7bd86eafa1ab87b6db60456d78427adcdc:action", "state_id": "b3788da5c3f0b49a2c59b56a939d653a64af4d0ef1bf231b8eca7e855980c6a0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4609375, -6.52734375, 3.578125, -1.30859375], "student_probs": [0.006389934103935957, 4.028877810924314e-05, 0.9861283302307129, 0.007441465277224779], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8a40c6d3adf0734783f0128e7c73f79bc5deeccf9cb7aef5fc92a546b79f8bcc:action", "state_id": "572f69f2c4c02d378ea559388ed7dd751b1077a3292ba9cf93915a28557a3ba2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.38671875, -6.0703125, 3.00390625, -1.37890625], "student_probs": [0.012090739794075489, 0.0001117876818170771, 0.9756119251251221, 0.012185568921267986], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9e195c5b923cf11c8e521a330c85270ce0e76f84731ff04e1ddd559d6fad175d:action", "state_id": "817abfb754d6c8b0e6d5ba82dfbf79b4cb2cd256e05137554725787b9fce4b56", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.09765625, -2.1484375, 3.064453125, 0.296875], "student_probs": [0.03812037408351898, 0.004903589840978384, 0.9004172086715698, 0.056558758020401], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7abf809d5b312b71efdd8e9f3b499e93ea65645b722278be4e055f15ca6738cd:action", "state_id": "945f83ab56df1b1f8451e0b9608e32f427ed863183c9c88c70927e2cbd6490e5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.30078125, -4.68359375, 2.515625, -0.10546875], "student_probs": [0.05278479680418968, 0.0006592916906811297, 0.8823859095573425, 0.0641699954867363], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f23b39c3305a39b2729299c055600219eeb8703da258d0f192791bb236dd37e9:action", "state_id": "60b2e0f15e5efd78ce8374f174f1f1197b917e19d913caa28e1d94d7c4b4dcd5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.63531494140625, -2.8828125, -4.046875, -0.11328125], "student_probs": [0.3540944755077362, 0.0374147929251194, 0.011681468226015568, 0.5968092083930969], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "08a240ff6f9b7aa4a15f9631bd5b2103791e2673e6b64c54f8629e117a66b28b:action", "state_id": "a0af13c04584f84b6e686a0a7f9b45a8b3bb902fe0e33e3667fe9abb22bde26e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9033203125, -6.13671875, 3.68359375, -0.87890625], "student_probs": [0.009977948851883411, 5.323597360984422e-05, 0.9797443151473999, 0.010224550031125546], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cb692d57e756fe81b2d837ac2212b378cbefc8f548d9ee7123d5e0046bee30ea:action", "state_id": "b4a90d26759c11d40ad87a4de2f6c9f6c6e6f1d2f2fc668c19302b0ee29958d9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3125, -5.890625, 3.1875, -0.974609375], "student_probs": [0.010819070041179657, 0.00011115666711702943, 0.973901629447937, 0.015168196521699429], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "80e06bb7489073f83bee51cb2655083ec79c694d6e2cf77db9e5abb97cc0daff:action", "state_id": "23eed89bd82bf2a62fccd807ee20402dfebed4a7872257f9113053a6f645e813", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.189453125, -2.01171875, 3.78125, 0.5078125], "student_probs": [0.017796216532588005, 0.002876920159906149, 0.9435874819755554, 0.03573932126164436], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d02800191a6b18d634deddbd84a5a22ae96de4d13e1abd6ebbb617fb1370efd5:action", "state_id": "d0725c9b5a6bb186c1b66a5490c5cc7f74aa92f38a58f416cfa542e8866b7591", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.03125, -2.28125, 2.876953125, 0.71484375], "student_probs": [0.0464295893907547, 0.004893642850220203, 0.850768506526947, 0.09790823608636856], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "59b1a78a369fccd4b9c30a2f32264920afa6aa13c07658c414eb84df55d4a9ae:action", "state_id": "1ee61454ca5b7b0ac5c0b321b243283471ba1e28ccbc2479b244dbfedb3a78da", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.0234375, -0.9765625, 1.3046875, 0.515625], "student_probs": [0.14547793567180634, 0.056086745113134384, 0.5490280389785767, 0.24940723180770874], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4e43d32f7de1825ea655ae3860cba1e4e038c99995990623551b6f7582a37055:action", "state_id": "02ea66ee59544d65e0c8c8e270bc026cf15a5d0498045c81eee419c6fb2dbea4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5009765625, -0.6728515625, 0.978515625, 0.359375], "student_probs": [0.11632246524095535, 0.09795334190130234, 0.5107388496398926, 0.27498531341552734], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5f3c1afd18dd9e02c1b16aee361f3c106809364c3a2a244416572dbcf33b4b18:action", "state_id": "65bc8eb0389185f677d64e72e66697988ca615fc0f6b243f89d5f1289aacba39", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4130859375, -0.693359375, -3.28515625, 0.4609375], "student_probs": [0.23760347068309784, 0.17952772974967957, 0.013443998992443085, 0.5694247484207153], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a6f90adc3f8bdda6c48df9794092f1e222339496e0c77e8a151eed958c8727ee:action", "state_id": "9e00d5c2bc16ce59e3336f64dd952f01acc9c25994d2ba418fafba1178214db5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.539306640625, -0.955078125, -3.484375, 0.04296875], "student_probs": [0.28550735116004944, 0.18838660418987274, 0.015017248690128326, 0.5110887289047241], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6a9b4d41472962f9df97b498bf46209131acc068ee9c98b4f633f6190791cf19:action", "state_id": "f2a4bda2617b9779e2131f254bebf656ea22ba6e6eee10f8da358c5db82d3dc4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.539306640625, -0.9609375, -3.28515625, -0.24609375], "student_probs": [0.32670149207115173, 0.21430838108062744, 0.02097219042479992, 0.43801790475845337], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "41730254d08b6913b52b1e03aea0bd9f1e3de73768fbee2a956a3323df2328b5:action", "state_id": "c51b0d85e86ac48c953087eca57b0452e572a9557cbecb11bee45eaa1863aded", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.810546875, -1.037109375, -4.109375, -0.1015625], "student_probs": [0.2586570680141449, 0.2062194049358368, 0.009551279246807098, 0.5255721807479858], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f266e6af741fd21529ed1356395c817f9765b503787a84dcceb27b2da5abaad6:action", "state_id": "6c1d459aaa581a290aba0297467272c67dcfcfcf8a6ca32de999e33168432a99", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8134765625, -1.140625, -3.62109375, -0.17578125], "student_probs": [0.2722243666648865, 0.19626742601394653, 0.01642836444079876, 0.5150798559188843], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0bbf22e91dec3f6be2edecb427f77f30a2017934cd035fb39d8fa79d21bccf8a:action", "state_id": "78dec879bf5884ee3cdc3933c4522bf52371dc12915a739934c687ccd39aec15", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.76171875, -0.896484375, -3.8671875, -0.302734375], "student_probs": [0.2856171429157257, 0.24960675835609436, 0.01279665157198906, 0.4519794285297394], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e15b3b8a8e81d83dd3a72afa93b6438efa9b35af0c8827ce98aa31e9d46dab4b:action", "state_id": "166a7be08329b9ef7442c38b224fc09aeeb07ba1ea2a2924ceb351e97c8ba681", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7626953125, -0.861328125, -3.640625, 0.1640625], "student_probs": [0.22278504073619843, 0.20186004042625427, 0.012531903572380543, 0.5628229975700378], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "83249f41fd6e5fc073cec0826abe3aa4f95c9849efc946333cca3ddef953d6c2:action", "state_id": "03b0227d909060ad855ab8a17ce27a66777640a2f2475d621f031df5ac497a5b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6009521484375, -0.90625, -3.38671875, 0.046875], "student_probs": [0.2695440649986267, 0.19862806797027588, 0.016625959426164627, 0.515201985836029], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "08df7590682550a24846371325087aace306e43ae688fc2e0bdaa37091708ed3:action", "state_id": "2bf5a4e4ac7c8a8bcb3207152e7b0ff5a1ecc842974a7171050e36c1add24edc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.64453125, -5.3828125, -7.6640625, -2.4921875], "student_probs": [0.9559468030929565, 0.0023056406062096357, 0.00023553601931780577, 0.04151204228401184], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cd7d247e0501f8cd310eb0806b1cae27a30a61628853c7754aae4fedc69eec21:action", "state_id": "0cecb01aaf9fe913df9fe0165f2b92f03b4db2191abae2838117a401130fe7c0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.015625, -4.2421875, -6.6328125, -1.734375], "student_probs": [0.8408849835395813, 0.01190123800188303, 0.0010898252949118614, 0.14612390100955963], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5b1538a6f439fddd0fd7ff1967d28e2db834c35ff6e034b34bf33913672132e5:action", "state_id": "f3b48da7f4c50dbfd76672e13e39e433baa6d66f2f39f952e18b9e2a6814cc13", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.326171875, -2.390625, -4.05859375, 0.3125], "student_probs": [0.32843026518821716, 0.041673749685287476, 0.007860912010073662, 0.622035026550293], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "22f1f5f9f0b838206ce89a3cbfd07d107ae0b96bb0df7e72595f002f643e0477:action", "state_id": "3920986da294b40143a28a4b2d6a03053e49fb6c57eee7a7f401f7ccc81c60ae", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.568359375, -6.4375, 3.375, -1.25], "student_probs": [0.007011485286056995, 5.384794349083677e-05, 0.9832947850227356, 0.009639882482588291], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5e0959aa4e2ec64abd4d6ca2f4d14aea44934a25a1b772e4a21d7787e7309e3a:action", "state_id": "386fce8f5992462b023179d246bc10b5d2957a9ec4dbaf337248987ca519697c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1484375, -5.984375, 2.78515625, -1.15234375], "student_probs": [0.018834413960576057, 0.00014953096979297698, 0.9622550010681152, 0.018760984763503075], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2ade727e688cbe853334c38589737a5b4b08858c664ffc71c321cd1e818b22ac:action", "state_id": "8b27e7929a53af8c9e823e6fa18ff43b27f3a84bfda95f66807ac7ad9393f8db", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.314453125, -1.00390625, -3.16015625, 0.39453125], "student_probs": [0.2784065008163452, 0.1397184282541275, 0.016173582524061203, 0.5657015442848206], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "72fd2e39bc903b0a40cc3f8444887071df4c3ef3e6943170777664c5ab9c4c5d:action", "state_id": "6fcfd9aba9be0de311c08e0b166c1751bc05022e30c8bbeab70139426232e5e7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.2421875, -6.0, -8.0625, -4.66796875], "student_probs": [0.7623024582862854, 0.04835312440991402, 0.006147409789264202, 0.1831970065832138], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e4e48f5c4787f4ba96cfc5994359cc649e419afdfecb8238a670f3874a8c64ef:action", "state_id": "b9c4473d538730730a02c0a30ad7db0441a20d3e2f99a66a1e7a7498c9829bae", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.6640625, -4.4921875, -6.5234375, -3.50390625], "student_probs": [0.6197423338890076, 0.09960165619850159, 0.013064894825220108, 0.26759108901023865], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0883ce15fbbfef5a64ca13c8d9defa3524e9fb06eb12b28091908c8857225460:action", "state_id": "445aae53aa70acd414293169dcff4f29730991d5845c807443b93706cd3b1f14", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.890625, -2.4296875, -4.0078125, -0.841796875], "student_probs": [0.9249829053878784, 0.012298321351408958, 0.0025379019789397717, 0.060180798172950745], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8d6d99d4f087f4813539395b9677e51796af74cb9d1d126f79128811c2418a70:action", "state_id": "3cfbb8fab20dc5f0f3f82ac29f90733cb40b24c44c4c4e8c41bf4547d144a58a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.44921875, -1.69140625, -3.48828125, -0.0703125], "student_probs": [0.357485294342041, 0.10322455316781998, 0.017116308212280273, 0.5221738219261169], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "50a756df0ebce464a259b42af61e35ae658e4b4f52392647b4343e662d3818bc:action", "state_id": "647eb1942f4e41bd61081d213c4376a1c13ee8df160f60eb22b865a71293bfcc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.2421875, -6.0, -8.0625, -4.66796875], "student_probs": [0.7623024582862854, 0.04835312440991402, 0.006147409789264202, 0.1831970065832138], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "aaf55d18bf05fcc900d47dd2e3a7cfba39e0089a6956dc2c4e10f4a393b8c1c2:action", "state_id": "6ccdfd5da81ae42479a2e4aa0a450d4fec0b541814933d8c12f48b977aac982b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.6640625, -4.4921875, -6.5234375, -3.50390625], "student_probs": [0.6197423338890076, 0.09960165619850159, 0.013064894825220108, 0.26759108901023865], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2dcab526a9734efa30ce274f342efc2511b592c0be22fac49061b2dacc416701:action", "state_id": "b546eb32c57bfacf81763586d005955b0dad07cb23e99c3b6ee1a59804c0de71", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.41015625, -2.53515625, -4.484375, -0.63671875], "student_probs": [0.7086639404296875, 0.037265535444021225, 0.0053060632199049, 0.24876445531845093], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d10ce76360e68acba4bdeeced391da99fbfbe711d8048c00b76260742fc1fa2f:action", "state_id": "a6a959f811d7f017d0151083b2a314999712fa0f08780d376d31079c7327b3b7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.197265625, -1.73046875, -3.35546875, 0.39453125], "student_probs": [0.32620275020599365, 0.0704086422920227, 0.013864283449947834, 0.5895243287086487], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e5d5a82868df63b9bd1e1220d1ba3a6707e533152922154daac2ffeb7674cbab:action", "state_id": "c122513c2b0c576ebf98bb5f063fe5ec40416eaaa47c308a155e30ff798045a6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.42578125, -5.6328125, -7.78125, -2.89453125], "student_probs": [0.9626937508583069, 0.0022504758089780807, 0.00026255467673763633, 0.03479323163628578], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2b9586444d37f0016503bbcfd716fd13661c53742e9080fafeae618710690fcc:action", "state_id": "d217414e0a71fba1e19ac656e2115969e0e9c5993223cbd9f3c82623ab4dc056", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.16796875, -4.56640625, -6.2265625, -2.08984375], "student_probs": [0.9593982696533203, 0.003101640846580267, 0.0005896506481803954, 0.036910418421030045], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "351028a5a9b64cc2d3401d2a164d923f77bf49e142adc3e37311e0ce46676d60:action", "state_id": "103c2aa254c9583f9fb6f83aa9113b8f97ae0392f382c45f9f3f809d351b4e5c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7705078125, -2.44140625, -4.3515625, -0.1328125], "student_probs": [0.32174623012542725, 0.06051339581608772, 0.008959447033703327, 0.6087809205055237], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f27a8fce5ea325cb3c080cc665ac4297b1870f58c468025256fe823f04ff899b:action", "state_id": "8586de3412a9bfe161c4afdecc8a7d28f9cdffc0a9f3f603d2d82633ac166507", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.39453125, -5.734375, -7.9140625, -4.62109375], "student_probs": [0.7140123844146729, 0.06878987699747086, 0.007778543047606945, 0.20941917598247528], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2cc1ff841dceed97859c07f1595a48ca4e0e8aeba67f650786ead05cebf77e03:action", "state_id": "33af3f6856aa55722d85db3ab848316e791920bed2b9e1bbdd8fcc35320f8be7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.87890625, -4.1484375, -6.3203125, -3.2734375], "student_probs": [0.5032766461372375, 0.14140227437019348, 0.0161147303879261, 0.33920639753341675], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "51c90906e56f18ac8a9995cab940efcab69a7ed163f833af6965657c91b21c84:action", "state_id": "bbd05ebf834b64b928f0d707cdbd1193bb6b5ee5acc555fa0001b61e46de18c5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.81640625, -2.78125, -4.859375, -2.05859375], "student_probs": [0.45174363255500793, 0.17213362455368042, 0.02154504880309105, 0.3545776605606079], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "705ba351706c83775a8ab42331cb71858b670ebce6a1ad7e111fa08497ccced2:action", "state_id": "00e22287aef07d1a27c5c17f7f08afc1fea8644d9d0ec67998b579ca02506cf2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.38671875, -1.72265625, -3.78125, -0.3984375], "student_probs": [0.6278071403503418, 0.07616164535284042, 0.009720764122903347, 0.28631046414375305], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c9034084e92941f1fc54ddf082447ff37cfc09bf687035ac09ee0560f0d260f7:action", "state_id": "8d0b086a3972a83c9e69ecbe117c4f2b466394b6c3c964774c122f115cca1232", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0859375, -0.99609375, -2.91796875, 0.37890625], "student_probs": [0.36644798517227173, 0.12419157475233078, 0.01817324198782444, 0.4911872148513794], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "693546ab2dd2c84aefd1b802b970c048a456972cc00609b5b1dd60c2836dd192:action", "state_id": "378448987f9aa7eb12030c55a286a5356ae22a560925648a10f13b1cf0dcdf28", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.455078125, -0.96484375, -3.19140625, 0.36328125], "student_probs": [0.25430914759635925, 0.15274730324745178, 0.01648123562335968, 0.5764623284339905], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "aef8ff624d5c94aaf1ca16495e16ac28142c3ec483fb2cfb5223a9191af63366:action", "state_id": "8575159fc53e15c887335fa15cfba82d19e38cdb27aa7d76b2f5ce792f7f85c9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.9921875, -5.96875, -8.0859375, -4.41015625], "student_probs": [0.769640326499939, 0.039226822555065155, 0.004721720702946186, 0.18641112744808197], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "14c4f19577a6896894afbbed15951130c01a7fd226e84a0825c0e8ad80364cb4:action", "state_id": "e904d7f989b1290289438b871a98e7cd7ea14ed9e6d88aa2a98294d637b35cfb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.5859375, -4.53125, -6.6640625, -3.49609375], "student_probs": [0.6400642395019531, 0.0914924144744873, 0.010842174291610718, 0.2576011121273041], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "297c315242ef33da97be5a1889fd82656f5ab40d65ab2f2dd61a0f089eae8420:action", "state_id": "ae026777ccea22dccf080fe648a27f7e42f43d4d185a28e80391d40a319e067a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3671875, -2.51953125, -4.5859375, -0.4853515625], "student_probs": [0.6715234518051147, 0.037443388253450394, 0.004741833545267582, 0.28629130125045776], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b6b44b75b4f164861484e12fb304133f47d3661f9e99920b395a393b0c62eb2c:action", "state_id": "9cae42d255b1d1ba405949c5578b98850f73a050ad6834725d549ad0219ebf93", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.111328125, -1.69140625, -3.3984375, 0.4921875], "student_probs": [0.32553601264953613, 0.06704707443714142, 0.01216257642954588, 0.5952543020248413], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "119ddc700d1b4649238230af75df6dceb2808c59954776b32ed1631ecbdd5008:action", "state_id": "f75b7cc2f44377993d156499069caaf39603c936ff304560b9c765471b3a08e7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.19140625, -5.1484375, -7.2109375, -1.34765625], "student_probs": [0.7560910582542419, 0.005318176932632923, 0.000676130352076143, 0.23791459202766418], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b2b8d7841a751005f5c9128ab702173990dce061037b1df6eb7df5cf624e0251:action", "state_id": "b1d2409710866f04e9d3ca508bd274bdf739a938f71c5f5a9a1db0daf280ce2a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.98046875, -3.6953125, -5.671875, -0.4365234375], "student_probs": [0.35737520456314087, 0.023663707077503204, 0.003278480377048254, 0.6156826019287109], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a1fe2e514a2d604ef368936a97181869be4ea8e5fa710e938cf7cb55ee80e74c:action", "state_id": "34374e6b63d8d093284ee72f8071d53846bf96ec4313e256b315e5af0b7898e7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.51171875, -6.546875, 3.515625, -1.28125], "student_probs": [0.006460240110754967, 4.2025036236736923e-05, 0.9853631258010864, 0.008134670555591583], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "66725daa20807f9b362091e2ed8bf0736cad1c640a927c4416dfdc4a5ffc33e8:action", "state_id": "f36f491c7aeca3ed5e656f2ad4b94e1d5fb29ed5584d7df37287b3789ed6332d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.501953125, -6.2890625, 2.984375, -1.45703125], "student_probs": [0.01100726705044508, 9.176230378216133e-05, 0.9773880839347839, 0.01151300873607397], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "017465de64fd38cda8a39995c021e1320811cf7f82a9c2db74495a990d9f96a0:action", "state_id": "64e3a17a8e269b33ed2e0dd785aa710286cad97fd62e17c6c6143e2ea8116f24", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.09765625, -2.02734375, 3.25390625, 0.36328125], "student_probs": [0.031971294432878494, 0.004642026033252478, 0.9126942753791809, 0.050692398101091385], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "18ed0ff3929587ea815e2a9dc18163337fcf38af00bee7a83dd8956c29935a1f:action", "state_id": "732877bcf1f09ad0ad29edf2fe9100d8246d633500f4cb3065123ed3d9bea467", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.1484375, -0.516845703125, -2.57421875, 1.12109375], "student_probs": [0.2366982251405716, 0.12169316411018372, 0.015551074407994747, 0.626057505607605], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "16fa53fd1e0368a3b962057dec18a076a1fe9747bfb843a5860afa824d1469c7:action", "state_id": "22e78a2bf29addaa49d517cabd263708cdc02660902e2f830c1fffa947a2dd5b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1796875, -6.140625, -8.0390625, -3.9609375], "student_probs": [0.840106189250946, 0.016000032424926758, 0.0023968450259417295, 0.14149697124958038], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "32e6eeccca864c6df0082480cedd689c4e2b0d6be7d9ec6844bdd591b6ffba26:action", "state_id": "050ebc8e1c04b1d0c968a603fcef71fe4bee5005225344bf4eab800faa686005", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.4609375, -4.203125, -6.5625, -2.50390625], "student_probs": [0.9417195320129395, 0.008878610096871853, 0.0008388444548472762, 0.04856308922171593], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "839dce13ba1d34a0a7b5791b8702ad5797f8f2b60da52c5b926083b13d047710:action", "state_id": "95c6a5a73a2a7b9ab005cf321c6057b2bf21c26670d04d15c11203d7790a583d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.447265625, -2.30859375, -4.1328125, 0.05078125], "student_probs": [0.35385027527809143, 0.05501169338822365, 0.008875787258148193, 0.5822621583938599], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9dfc6f4fbe309b0de730dddc95547f7ac16d05cedbde6b3aaa0fc45b34e761b4:action", "state_id": "8a69097d1599909d7c8ec85a8858009464f1d6946b7847a79307026df7422d4c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.140625, -6.234375, 3.65234375, -0.998046875], "student_probs": [0.008142100647091866, 4.995154813514091e-05, 0.9824181199073792, 0.009389822371304035], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5e7d7563479038c303891745d593238c15f7c2ac5cd4ff7d423426f95f4f6576:action", "state_id": "8b7f0137dd7b2c5913136332c32bd73c04c6335aeec444e3ad3981a892ff1254", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -5.25, 3.11328125, -0.55029296875], "student_probs": [0.015070337802171707, 0.00022396714484784752, 0.9600883722305298, 0.02461734227836132], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c533489d7c717feb4c5533ac31efac0e5976f99cb95ac92f4cd8c3fa5c4e0a1b:action", "state_id": "0e62d32f9f2907f90d8ca209ab77266738054ceb33096fb3eca9213053a4715e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.34765625, -2.21875, 3.611328125, 0.37109375], "student_probs": [0.01798240840435028, 0.002768484875559807, 0.9423515200614929, 0.036897506564855576], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e843bec7da3cbd0f145324deb808f2c0bfdcb3e9e57c4665fd6ef7f413c6e565:action", "state_id": "7546d437c082a412161f9ac85b6c509277d791d3c604ee455952f4b8e490cef5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.68359375, -2.28125, 2.8984375, 0.58203125], "student_probs": [0.08996874839067459, 0.00463955570012331, 0.8241117000579834, 0.08127998560667038], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "18ab53bf046c7ff04a7c276a6f00d03e0fefc5d4ba451ffc86faa7b71ce32eca:action", "state_id": "75c887242c5fcef3c4e257fa1d8c741fe84d6886ffc4f68a22823357bbac1165", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.125, -0.6611328125, -2.484375, 0.5859375], "student_probs": [0.32105371356010437, 0.14627313613891602, 0.02362329699099064, 0.5090498328208923], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9e8ab796083dfa2386eb74a33b79cf1421bbe6d0dc6902fa5f17078822dc890f:action", "state_id": "faa2665681d6626ebe9b220c56734a5e6ae7307369ea0874e326db090f69c157", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.310546875, -0.94921875, -2.3359375, 0.671875], "student_probs": [0.23090142011642456, 0.12191437929868698, 0.030465662479400635, 0.6167185306549072], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9e71a64ee5b87ad919f35ff07bddac26c788cf7b1fc8649df974b2d1d819cd6d:action", "state_id": "7455119495e3919c9d45edad7a7152da25bdb667173a91f2939252fd1fe6b6f8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7900390625, -0.77734375, -3.8828125, -0.07421875], "student_probs": [0.2436637431383133, 0.24677684903144836, 0.01105646975338459, 0.4985029399394989], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8079bac13cc45d7fca4b75078b0bd8f712a8378a4e237ea6ceedeb281bb8dbde:action", "state_id": "77d1c4bc21017d6bc7a6197353ad99f2a648ec264da4eced2e4e8c4a6f9affff", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.14453125, -6.21875, 3.65234375, -0.998046875], "student_probs": [0.008110609836876392, 5.0739748985506594e-05, 0.9824486374855042, 0.00939011387526989], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6789a2fa624caa3500e06e12a79438220a442e386988a6932cba3c5341151fe7:action", "state_id": "f69c393cc14b5010b8edd3aa3a966e03b7cce0d04bbee1848b7dbc55cdfd6f64", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -5.2109375, 3.11328125, -0.55712890625], "student_probs": [0.01524769701063633, 0.00023288461670745164, 0.9600703716278076, 0.024449173361063004], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2c71f49769d21850216fb8fe14ddb5c17bbdbff8f93ba7105e97e4cdad9b52bf:action", "state_id": "37504feef3d61ec67141484354ac7fb348f25fcecece2bf6db22cf58582640fc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.341796875, -2.2265625, 3.611328125, 0.3828125], "student_probs": [0.018078701570630074, 0.0027455156669020653, 0.9418627023696899, 0.037313077598810196], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5cb5dec11f05d8f199f5751426304f9b62fe927bf3d81ff0a8a292987cff24a8:action", "state_id": "5b53cd04b0f30ee6e193ca4d19f4b3ec60ee03eb5ba76897feb3a2193f035767", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.69921875, -2.28515625, 2.9609375, 0.60546875], "student_probs": [0.08650028705596924, 0.004374414216727018, 0.8303658962249756, 0.07875940203666687], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "26b4f0c47ad8be318a1d7ca670e68cbe786ae0b6160252c332b5872fb72a5e11:action", "state_id": "dcb38000166f612842142404075168581c2aa7cf4ae6d51c1e42a790cdafb038", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.125, -0.65576171875, -2.46875, 0.5859375], "student_probs": [0.3206818103790283, 0.1468905508518219, 0.02396751567721367, 0.5084601640701294], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7e7d4f893c005dc86d37c9e947049b4a96030ff493f57be2f2eaea4e0ffccd1b:action", "state_id": "77a9a52ed244c70c528759002f6a0926ed3d53e92124fdd05463f9739a40cced", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3251953125, -0.927734375, -2.3125, 0.66796875], "student_probs": [0.22808930277824402, 0.12486062943935394, 0.03126291558146477, 0.6157870888710022], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0ba62142f7648d963c52179ebf38c314127dae6aaf88dc87444e48ef029d3614:action", "state_id": "14f1845c06784e4b332e50287c1b9964148f4f5b9be8faac60609634f65d16a1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.671875, -0.9765625, -2.72265625, 0.0078125], "student_probs": [0.2604675889015198, 0.19205676019191742, 0.03350508585572243, 0.5139705538749695], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "78d8409580dfed0012e117106f0dae7b35eabe7920785f99d122ec8d9ac40de9:action", "state_id": "63f09b2b310611e6420726af842e1366edf1454ecfbbef78311cf98f943154ec", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8388671875, -1.32421875, -3.3671875, -0.2890625], "student_probs": [0.29169926047325134, 0.17953528463840485, 0.02327554114162922, 0.5054898858070374], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "33c90eab8e23504df2509fa67af9b93b22fea42235ab7f79bc70ab038b550fc8:action", "state_id": "8b2fc7d39e2515e0ba81f420839dd929c738b555d64e0200744c7d219e9fe709", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.9609375, -5.96875, -8.0625, -4.375], "student_probs": [0.7700362205505371, 0.038039498031139374, 0.004687385633587837, 0.1872369796037674], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3bd9efa6229b286f94d41196384b2ce0aa560a51e43e91ffec0384f5ab103ab6:action", "state_id": "ef80c560137b7c1097df50fed57f5bce4177c4a3e156bdae0499edaadd0a13e9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.6171875, -4.609375, -6.7265625, -3.5234375], "student_probs": [0.6423221230506897, 0.0876106396317482, 0.01054566539824009, 0.25952160358428955], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1818fc457e89fb96392d4ffa791cfabf8e88b95d662d94bcbff1017379c88b0e:action", "state_id": "d5ff485fcbffdbe4b49814c413b6a5ba04638c452ef73b8a8ae313ca69c3ca5e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.375, -2.53515625, -4.5859375, -0.5], "student_probs": [0.6764363646507263, 0.03684360533952713, 0.0047393543645739555, 0.28198063373565674], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "810515236d640bc505de88eea94a41e9498133d5dc84a6abe627fa4ea5d7a68e:action", "state_id": "1654241e3ac4e4e8002a4c7c657ce8cef9d24766e981478639615a72f6dc9a78", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.111328125, -1.6875, -3.3828125, 0.4921875], "student_probs": [0.3253883123397827, 0.06727895140647888, 0.012348503805696964, 0.594984233379364], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bbf136fdec004bfbab867309b67d7b535acdafdbae22bb33f8d1cad8104111d1:action", "state_id": "95d36313b3d03aa30c7d0768d4c2c4e5c9c1cf62fe7c0db93891f8443a3f6a7b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.296875, -4.6953125, 2.66015625, -0.1171875], "student_probs": [0.04662024974822998, 0.0005732676945626736, 0.8970093131065369, 0.05579713359475136], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3a2c3b7a777a4109294b688d44f572a2991bf582bb2f473246d064a9753336a8:action", "state_id": "1b3fee9a3ae55437d1ae348832b2dcfe57979aaa97e286029e9090838de47825", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.73486328125, -2.88671875, -4.484375, -0.09765625], "student_probs": [0.3299253284931183, 0.038359832018613815, 0.007762889843434095, 0.6239519715309143], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4a9cdb0f43a3f3ec46bf55d9a45affbdaa2172cbe3fcaa32b91d8462dd877842:action", "state_id": "6a368accf4dfb4f9138659a7d542f2651529bd50e60a3bd43983771238eca8e7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1875, -6.09375, 3.62109375, -1.1328125], "student_probs": [0.008024215698242188, 5.938070171396248e-05, 0.9834411144256592, 0.008475261740386486], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ca47a66d52f4e605c8b6a65653ccb1e8605126821a82fff267faced978ba0693:action", "state_id": "2ab9a8031f078df1970066c8f9d04fa4584eb6e25c9263c8cdd73ee1e6ed793e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.19921875, -5.53125, 3.125, -1.060546875], "student_probs": [0.012875251471996307, 0.00016919146582949907, 0.9721651673316956, 0.014790407381951809], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c0e02ad4a67cee0a9fb68590dce10dcdbdbb8fbb4fc620f5669857f431f0a5fe:action", "state_id": "37bbd5bf003e55f55d13076841b72dc5e65f84df4b0b3929f86789bd0a5d0528", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.11328125, -2.2109375, 3.28515625, 0.2890625], "student_probs": [0.030735764652490616, 0.003772623371332884, 0.9195315837860107, 0.0459599643945694], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b62512b81da826040478b256f2a0f8b942788e1bfe433d53f4bc569bb735d2a1:action", "state_id": "d8823d9accbe05c6f0865e734f96f3a1e64f13e753ec52f43e606ac48ed8d6bf", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.046875, -0.5670166015625, -2.79296875, 1.0390625], "student_probs": [0.2327311784029007, 0.12596352398395538, 0.013599597848951817, 0.6277057528495789], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fbbd4f805c902624da7416e49d642d8938698744cf6e4cf80427594090f168e7:action", "state_id": "cc02efa9bda98e827a3d936cf1d3344c8761f5b0ee87e20fa039989a607872e7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.146484375, -4.58203125, -6.578125, 0.15234375], "student_probs": [0.21269433200359344, 0.00685041444376111, 0.0009307313594035804, 0.779524564743042], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "33ffa57830040110519e0d095f2a1d235d952752d90b467f6d22b35fc0dc56eb:action", "state_id": "def9ed56abc6ce54b712cf990ccfa50c2f545c92a0b9a62eb35ed251527882f1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.51171875, -1.671875, 1.08984375, -0.9033203125], "student_probs": [0.058226577937603, 0.04960966110229492, 0.785173237323761, 0.1069905236363411], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3c8d8881dfe4f93ac0b445227135fc30993a7a962301bd1066e7b3f70aa28fc7:action", "state_id": "c5e8a31996230754705c3c6a2c30c51dcb6d5c220811c57d25266a21bef6aa7d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.86328125, -0.125, 1.544921875, 1.318359375], "student_probs": [0.2030196636915207, 0.07556714862585068, 0.4013940095901489, 0.3200192153453827], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4658bee94bf48fa2fcfc4f803d9e175e9f79f69839484342cd3ab072674f5836:action", "state_id": "bf82d553b17b1d534bb03c7e660d623c58ccd093c662c8af2da7766e15af3214", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.890625, -0.185546875, -2.62890625, -1.453125], "student_probs": [0.11725281924009323, 0.6451033353805542, 0.05603918433189392, 0.18160471320152283], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ec378bbea42158f05a341da9b599fa31feedc4d67b05243e9b0df64ca3afa0fc:action", "state_id": "b5eadc9666605b40a875fa4a805ab98a99be635e157856b5ff860f32cb15d932", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.80859375, -2.33203125, -2.75, -1.54296875], "student_probs": [0.30424681305885315, 0.18026027083396912, 0.11868026107549667, 0.39681264758110046], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e8bb8919d176f0a8584380fe62aca2bfebba581a74f3e95b33e1d8d10c07ce2e:action", "state_id": "64837ecbb4c59e1dc7952ab080b7855786ba724d311fd220adda0587d8bf3465", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.416015625, -0.6751708984375, 0.5859375, -0.3671875], "student_probs": [0.18033161759376526, 0.13916248083114624, 0.49115049839019775, 0.18935538828372955], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f4ea606ad92472c43ff2465e5a2183e4e677f322ccebc40b63bcfc61dd134cf1:action", "state_id": "28b9a0ab3b7a401a6ceef767032a1a41f2a7dc1645093dbe5cb6d279c2f07a07", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9296875, 0.15625, -2.046875, -1.177734375], "student_probs": [0.08290021866559982, 0.6675239205360413, 0.07373297959566116, 0.17584288120269775], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c8a58fc9c50f97b87b6f0842ae503f04607d6f8bc6facb6332f2597a2a5355c8:action", "state_id": "5f521ce766b21262b097fa2a854f0d33742beecf4717522f8dfc6e5079ec2435", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.83984375, 0.12890625, -2.421875, -1.123046875], "student_probs": [0.09286478906869888, 0.6650714874267578, 0.051889337599277496, 0.1901743859052658], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7e20947fe44761d508eb48b77bf12483c544f18f5f131b24a89c5cfcb2b75733:action", "state_id": "b6bec25cccf0f1bf0cb6542c9afd8e0ca10e6e49ec95f785a2a848f5b3306c97", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8828125, -1.94140625, -2.12109375, -1.5859375], "student_probs": [0.24529537558555603, 0.2313355654478073, 0.19328811764717102, 0.3300810158252716], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e5759197caf1c7541c0de603513a204bd2eac6e8748445e80bc06f97c55af195:action", "state_id": "533594d605a026135a85488f75c75a3d03d99f4197797f9b2b598778913aed91", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.5390625, -0.1328125, -3.19921875, -3.140625], "student_probs": [0.02937186509370804, 0.8856194019317627, 0.04125948250293732, 0.043749261647462845], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9516e62eec38ac2d18402becde4842a39f85ff83ca16590b7db9b4af68b6cc6c:action", "state_id": "a0bddc843f116d93af7984e46708eda370d776a21dade5b74fa36c1d5980dfd6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.359375, 0.21484375, -2.3828125, -1.77734375], "student_probs": [0.05921515077352524, 0.7769657373428345, 0.057843439280986786, 0.10597559064626694], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "68910e58ba370758f748e5a37decbcb46f07dd6e7f3552a3f1cfd0c98c028456:action", "state_id": "14055c0710db51c6ec54cfc277bd32a0acfb050d1a8c49ddfb135fee5553372c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.28515625, 1.24609375, -2.1640625, -1.41015625], "student_probs": [0.025843629613518715, 0.8829902410507202, 0.029170500114560127, 0.061995647847652435], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e9cf8820552e2e9358941a58572781c8326d01aca88bf8bcb4bed4a53e2be828:action", "state_id": "04c6273b707a72442bd86b304516e5bc051aa5a64583062eaf3209e91aae065e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.23046875, 1.34765625, -2.12890625, -1.21875], "student_probs": [0.024592041969299316, 0.8805510401725769, 0.027220910415053368, 0.06763608753681183], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ee122f56b8a561a3e3398cc5cb9507c95545f21ef5f319a6ad116c9d286ab790:action", "state_id": "7b23e40efd69442be2316256305074cf90e6ed24341a18b978e529c8bf48d20c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9453125, 2.0234375, -2.03125, -1.515625], "student_probs": [0.017739061266183853, 0.9387216567993164, 0.016278276219964027, 0.027260983362793922], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "95ddf49b94af02ac5e55cd60e657e3eab7f216173d007b6755c4b86a4e76b272:action", "state_id": "4e433878cb8d52917a75c12601d7075cf1606b09a4eda89abefc20a23d5a0ab2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1875, 1.98828125, -1.68359375, -1.123046875], "student_probs": [0.014155264012515545, 0.9213756322860718, 0.023429427295923233, 0.04103969410061836], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cf1759de66a552151b5a662c964c58d96ffbadbfcdb2679f833a98d969f0f572:action", "state_id": "480c90b6b32c78f79e0c3e2d1fe707c1c5695c1dd20f58cb3e0173e44e005696", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.40234375, 1.560546875, -1.19140625, -0.73779296875], "student_probs": [0.04249480366706848, 0.8224374651908875, 0.05247407779097557, 0.082593634724617], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e4bc05df9c0ea7bbd4f5b5a41d95a7bd941a725814615e2f23d60647e03296e0:action", "state_id": "68b9efcec43919d1d14e28c43172b68a30e51c352bf3c25e2056662a3ea6cb1f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.730194091796875, 1.97265625, -1.12890625, -0.244140625], "student_probs": [0.05488692596554756, 0.8190339803695679, 0.03683922067284584, 0.0892399400472641], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "851542dc0c83abcb06c934cd8dc4f170f1e204a3d1e5aed3f97fa00585c57c6e:action", "state_id": "7b6855d7f92cad850f2edb0cdb5752b59b96a039ed027a8eb09625d3c2ae93f7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, 1.58984375, -0.8896484375, -0.564453125], "student_probs": [0.05652201920747757, 0.7863820791244507, 0.06588762253522873, 0.09120830148458481], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ae5395869767c634e09c76f6397c8bcbee3f3ca54767fc7c084c8eb94cce6462:action", "state_id": "6d2a5ea2ddd9a8b614e308cadce5a911f24580f717686396930fd18f580fcc4c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.61328125, 1.73046875, -0.74609375, -0.6776123046875], "student_probs": [0.029193442314863205, 0.826908528804779, 0.06948643922805786, 0.07441169023513794], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b308e1944543302705827cbc49835fd827bca083591f3093c302de98813d1408:action", "state_id": "fc042534be6fb66d8aae427721e1f28a78c36d09ce9b3362dc53205607e0dfa5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.81640625, 1.640625, -1.37109375, -1.025390625], "student_probs": [0.02740531601011753, 0.8693695068359375, 0.042779091745615005, 0.060446131974458694], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8c218799e697c7a4c816f220275494ec46191b68c58f4e63dea4d638069c5d4b:action", "state_id": "71a68b34318eae49bb60e9e771560a56debc92fb67dedfeedaffdb6dcb478eb7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78125, 1.359375, -1.58984375, -0.90234375], "student_probs": [0.036052241921424866, 0.8334669470787048, 0.043657511472702026, 0.08682332932949066], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0fd2200ec4bb36a17bb288f4537a06b01d362f00ffab06aa1d4bb912cfccafe2:action", "state_id": "86de8f096c23fc8ddb9f8062774ef504bf62372b479455364e9a3a0dc2ec0183", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.828125, 1.0390625, -2.20703125, -1.33203125], "student_probs": [0.04781397432088852, 0.8409274816513062, 0.03273391351103783, 0.07852457463741302], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "58ccdda00633e3a3832e96362d8be4fc445390a1640e7bc2b199f4f77741cd53:action", "state_id": "2ab741e16ac480a796dd69f8f81d2bd2b3825262925e407d0d3698832148f83a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.35546875, 1.0, -2.4609375, -1.8984375], "student_probs": [0.03111550584435463, 0.8917403817176819, 0.028000926598906517, 0.04914315417408943], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2bb56ed2043ceb89d34c8931855102dea43f00e893f3297a3e92307da0289adc:action", "state_id": "669bc2c22aaabc7a4533fdb47a02d68df085e64a7de351891a9ca82a66fa5f04", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.796875, 0.61328125, -2.39453125, -1.28515625], "student_probs": [0.06966720521450043, 0.7757931351661682, 0.038323886692523956, 0.11621575802564621], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "45a1fc39dde257b7ffc0564475b3ba522e11efddeb14b414a7c530d7cf24e038:action", "state_id": "e9eb1c97acc9f12277c724a8d4cdf4c208bd1de00470dcc1b7a74ddf88f898e9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.16015625, 0.66796875, -2.03125, -0.8193359375], "student_probs": [0.11053629219532013, 0.6877799034118652, 0.046258725225925446, 0.15542513132095337], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "655bd63ca8e9770dfb4fc6f3a0c2d4d0151288abf1b022595d8a4f5593b75221:action", "state_id": "bb8fa6dbef54a029eaf1f17346324c8c05ed097ff480a6a82193127890d8cf1b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37109375, -2.0, -1.63671875, -1.017578125], "student_probs": [0.2685309052467346, 0.14317385852336884, 0.2058897763490677, 0.38240551948547363], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "eb548696b65e632131302ebd21fb321869f2056e4cd28b94bd51bd1f85fa2c28:action", "state_id": "6c3b4c37a7648235883a17b33deb86c5efc45002ae878da36c29c1597452ac27", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.89453125, -0.548095703125, 0.7109375, -1.119140625], "student_probs": [0.04865538701415062, 0.18701672554016113, 0.6586756706237793, 0.10565225780010223], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fe90c3894878c917f0333b14bcb31affe2a192d0248f6a43db3355f63792b737:action", "state_id": "cddbd3cc42ed99814151a430bf2f889008e7c156b85c608130d9e385593c7f4f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.67578125, 0.109375, 1.28515625, 1.291015625], "student_probs": [0.19022499024868011, 0.10796437412500381, 0.3498772978782654, 0.35193338990211487], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "168c2fb88b39e3f7f16d3fd847ed10189ea0036df5cdac4c74030f9ee6acc68f:action", "state_id": "f4b437835adf30054c1a1691b42390a5644188b4482374940660d947ce895d7b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.04296875, -0.3359375, -2.63671875, -1.5546875], "student_probs": [0.11501739919185638, 0.6340416669845581, 0.06351864337921143, 0.18742235004901886], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "453c034591e4912735480f4964e82940db477bb2df310e07dc861c17bc53d55f:action", "state_id": "d2cdd1e92d8a9c0cccc89b34dac95f913028041b13b9c478c8a2ddfe343cf9f9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.04296875, -0.314453125, -2.4765625, -1.72265625], "student_probs": [0.1154998317360878, 0.6505282521247864, 0.07486416399478912, 0.1591077446937561], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c7b7d3df036db6c5be902528c44cfd6caa4df6e3d5f11b37651d4a25d8974e1f:action", "state_id": "d2d99fa9ed53c35fd70059e5b1b7788380db9146e0f3a426d6f8856207c361a2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.22265625, -0.412109375, -2.21875, -1.9609375], "student_probs": [0.1061924546957016, 0.6492383480072021, 0.1066080778837204, 0.13796110451221466], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "75e499181babe1542ab9ef02d5600abd790fa503f11de48e0a04497221267fe7:action", "state_id": "807c527436155c64b6792edaed96eba841906e1817bbd879fe9f0fe30e75b73d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.12890625, 0.33203125, -2.23828125, -1.75], "student_probs": [0.06634436547756195, 0.7772766351699829, 0.05947070196270943, 0.09690828621387482], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0f2612cfe3c63c55b85b106dec9e4b8f0342e502e901c9d4fda55a4b7e7aa8ad:action", "state_id": "b36eabc74d9ab6b0935d0f24d80ec632af2a8f586907cd433b6faf574f2e5130", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3359375, -2.40625, -1.7421875, -1.55078125], "student_probs": [0.1684744507074356, 0.1570354551076889, 0.30506783723831177, 0.3694222867488861], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "af94ed53788c6296e67bfc4b512a2f6c326051bf8e8ad4bab58c7afb5f17eaac:action", "state_id": "5838952de5e1eaf85419c3508f2c1608370ad28be935034813afb0d8a3185c86", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.62109375, 1.083984375, 0.09765625, -0.690185546875], "student_probs": [0.04154575243592262, 0.6213368773460388, 0.2317236065864563, 0.10539376735687256], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fc8815f8f6199fd67e511cbf6f5df1f82d9b78919aea15ddd76a0506529106d1:action", "state_id": "2014b9cc22537e532345f488194b596863762c1fd3a66da92b13c05860b47946", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.64501953125, 0.625, 0.5078125, 0.4453125], "student_probs": [0.0934288501739502, 0.33269283175468445, 0.29590314626693726, 0.27797526121139526], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ccc46d0bf02c09662f7e7bef255aae1371a912b3d8f30046f738483047613535:action", "state_id": "14fb47e5d70acfc09ab275ff89d78a1e954655dc4ea4949f6e06fd32be51bb76", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.67578125, 1.20703125, 0.33203125, 0.36328125], "student_probs": [0.07611433416604996, 0.500220537185669, 0.2085229456424713, 0.2151421755552292], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bb557be7c4194ab3e8b6be5b17c43cc09973bc939ebd22ab98f188f69468ad94:action", "state_id": "c3ffd3f8d8403d3589581d57319f28b1eab6bfbd36b2a80fb01c5105d4f37678", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.84716796875, 1.51953125, 0.17578125, 0.09375], "student_probs": [0.058803264051675797, 0.6269686818122864, 0.16355456411838531, 0.150673508644104], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a0d5e3e668a8534bcb18a54deb0440328f153a1740c72623d1230bbe5b192c8a:action", "state_id": "2f95df0f98a7325afebef4e90c04f92a214c092b4d2a1bf5840eafdca2d60c14", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.177734375, 1.734375, -0.09765625, -0.19140625], "student_probs": [0.03996508568525314, 0.735180139541626, 0.11769356578588486, 0.10716120153665543], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "10f0af5e125f7242f9ffb881407632b488ff7c68b1268f80eae30dc8bc0ced15:action", "state_id": "b3c3789118dc7b27a20a3aaff90af20dabff925a7f51863cad145d0224e235d7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8154296875, 2.056640625, 0.07421875, 0.296875], "student_probs": [0.04140923544764519, 0.7318490147590637, 0.10080140829086304, 0.12594038248062134], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c3b4083e2833a57a579e3cd82fe6c1b0ec616aac90006cb5f7e208f162704c59:action", "state_id": "50c31029e936071b758bf29e5d2fda7ca7f9d2a0094ef7af8cc67f673735bbe9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.50390625, 1.177734375, 0.3046875, 0.515625], "student_probs": [0.08778852224349976, 0.47180765867233276, 0.19706320762634277, 0.24334058165550232], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ae4e84e16c9472ec60227ed7c8fc173edb95a8ae956b5f985683c18aa2af3dc2:action", "state_id": "87d0d00fb5225e8be4463b34612e8da1c79f71dce86a160347024a91f57136c3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.962890625, -3.015625, -0.532958984375, -0.330078125], "student_probs": [0.21985585987567902, 0.028225839138031006, 0.3379519581794739, 0.4139663279056549], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "67b696fc9be71a9b9dc1cc459080f57137784dd3e56a4fc6714ca504efc033c4:action", "state_id": "35f0a452768f88df9b0a21e664e935cb7047a80522964320f1f6cdf1bdec9a63", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9765625, 1.09765625, -0.4658203125, -0.923828125], "student_probs": [0.03330164775252342, 0.7204136848449707, 0.1508595198392868, 0.0954250618815422], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "17432f07a1d670103d50ebcf62ea0cd5745543432c09b1be3785f154de4374f2:action", "state_id": "ba1492ce74c761ea9644d6f8f21f6403592b2125b60e01950e6d62b1803b9431", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.080078125, 0.65625, 0.18359375, 0.4609375], "student_probs": [0.0671854168176651, 0.3813754618167877, 0.23772822320461273, 0.31371089816093445], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "276ede00c4686c958f4b65ac980b300ee5988a632832b5ec9ef145307d69180b:action", "state_id": "112537439502032d884886c6ceefd824dff905d876379c25682879afe5cb6941", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.82666015625, 1.322265625, 0.4921875, 0.63671875], "student_probs": [0.056704502552747726, 0.4862774610519409, 0.21202437579631805, 0.24499370157718658], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0d158c3586a17a20d51d679d999eb64c5b3154981c3deaee6366539a9d8eff04:action", "state_id": "810af90e453086de58677622cd313e5a0a5fc6cfeebe0e607801cc4f9753d7a9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.845703125, 1.759765625, 0.39453125, 0.4296875], "student_probs": [0.04635189101099968, 0.6274919509887695, 0.16021175682544708, 0.16594438254833221], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c24eec1640e2c9ef4d523b00b6c55723c283e40f78bb71e9a4178e70a8ccfb1b:action", "state_id": "78227e916422089c7dae1beffa6953a66aa942874b43a40831f515df6721c3b4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.234375, -3.69140625, -0.380859375, -0.545166015625], "student_probs": [0.1843075156211853, 0.015793120488524437, 0.4327331483364105, 0.3671661913394928], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "db47cd7c0bcfda1021b3990c2df2f5f18db34fbab36c340d72b7152ce2a00b93:action", "state_id": "6083235d6a9077c98527bc4df0687637dde816f74d468b4d1f65d80af907bb76", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.53515625, 1.2109375, 0.2421875, -0.4443359375], "student_probs": [0.03925803676247597, 0.6117048859596252, 0.2321769893169403, 0.11685998737812042], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6332bbb76cd0f983ebf5f68e0f958d94e1dd7e3321f69d5612c92c743303432b:action", "state_id": "512ed4e0003090981ce7584bf652357f1831f52fbbae4fa0f3f9e324525516c4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5947265625, 0.689453125, 0.666015625, 0.54296875], "student_probs": [0.0888153612613678, 0.32077479362487793, 0.3133440613746643, 0.27706578373908997], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3757694bdd5b0311ae069f823101d4eac4a019b84518a83443f6f480b49ec7d4:action", "state_id": "5a3b36b9d89b8e8027f697a714470e95c30af78d751a4729a94a83b933968f7a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.994140625, -3.66015625, -0.302734375, -0.4716796875], "student_probs": [0.2104278951883316, 0.014630777761340141, 0.42012372612953186, 0.35481762886047363], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b9a2e1f38e8dc8b35037f128a14eb03fde091dc77269be0d4b4e8b48a2560fc4:action", "state_id": "efa85f01ee4e3262e7046f3299005384b290c59ef46dc8e62a4c1896dced63df", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.45703125, 1.1796875, 0.2421875, -0.501953125], "student_probs": [0.043410610407590866, 0.6063289046287537, 0.2374418079853058, 0.11281868815422058], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5ca58916933b6689ef5f41fcdc7cada87272e292730613257f920d43cf89d050:action", "state_id": "294d3f1f01294aef5cfa0517099e4bc1fd2926dab75305bfb720453e41f8bcfa", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7283935546875, 0.73828125, 0.5546875, 0.5703125], "student_probs": [0.07932046055793762, 0.343838095664978, 0.2861674427986145, 0.2906739413738251], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3b782fca7da1077f0da4af299ca6bb96b16bd81dba8df67e13445e4d790e1bca:action", "state_id": "5d44dff3042c4e1eb9e252489f20c77edb1020e552391edd03adbc26ebea46d8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9375, -3.2890625, -0.234375, -0.3486328125], "student_probs": [0.20336687564849854, 0.019364647567272186, 0.4108123779296875, 0.36645612120628357], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "708e634ccf5d12e3871e356d865ecd207b5b1a21737f7b98badfcfd08e48d133:action", "state_id": "38708733c2a4547fd01b44cb7b7ac064e7008628fa2dd3f778bbba0068aeb573", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9150390625, -0.9453125, 0.697265625, -0.03125], "student_probs": [0.10633109509944916, 0.1031603217124939, 0.5331817865371704, 0.25732678174972534], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ec642887631c168417c237325b502c639e98a87014f63049413fd8c2c5fa741c:action", "state_id": "74fa4c03b0ab15db4ac145689302441533fb65fb8ebd5d9e9ee28c64783334f8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.2109375, -0.404296875, 1.025390625, 0.912109375], "student_probs": [0.1719817817211151, 0.09295859932899475, 0.38832464814186096, 0.346734881401062], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f48c3e7d06370fd95220d8937a705aee7f8f52257865743360febce8f3461cd8:action", "state_id": "5eeb3c0dbdb0a2b2f772aef7d8183f542ec4b90322e7bc150c91fb76425944c9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3359375, -3.08984375, 0.046875, -0.0546875], "student_probs": [0.25941237807273865, 0.016519024968147278, 0.3804031014442444, 0.34366557002067566], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e23f0ac772a0479e8c36639fd9a02249fc9b68f33acb55bdb5c49985e03c9017:action", "state_id": "190dcfdcbbab54c3a908c0ef01f226b091a29ae454aad7976368610c0cfa39b7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8955078125, -0.8388671875, 0.689453125, 0.06640625], "student_probs": [0.10466737300157547, 0.11076691001653671, 0.5106826424598694, 0.2738831341266632], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0ac5b7f8d14a92f9a803e79c3b20ba8451aa5f0d3aedb5b2ac4df6c09cf4823a:action", "state_id": "908bf5d51a7fd1c9c73a69f2f79cc394964df19302850ec1c81f389eea334cf1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.263671875, -0.427734375, -1.18359375, -0.77020263671875], "student_probs": [0.16588324308395386, 0.38268861174583435, 0.1797132045030594, 0.2717148959636688], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9e04499a6b9dfff6b15760563bd06b17efcabf6ace988f321f63e040f6924655:action", "state_id": "a8424cf9c9883775e60ca317ce4a3a624d5361c7b4218a88e7abefe626ab19bf", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.21875, -2.17578125, -0.8876953125, -0.986328125], "student_probs": [0.24764002859592438, 0.09510152041912079, 0.34482288360595703, 0.312435507774353], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a853850ce53a528d8624e6006775a84eafde90614360541dd8185fb3ee7b7655:action", "state_id": "99fd347e457f402e1256f335f027dab19798f5a4ac4d1cc4cb47cde6047e8700", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.31640625, 0.22265625, -2.85546875, -2.88671875], "student_probs": [0.02593565359711647, 0.8930844068527222, 0.04112252965569496, 0.03985732048749924], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "88ad6c034df007cce8b93e3f8b897d6a1da4718bb75d12b830aec6ab02dcc4bc:action", "state_id": "089a91a7b840dcdae649464e3667aa5d4d9b39d2cba07b860fda8dc31c339a32", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.53125, 0.25, -2.08203125, -1.265625], "student_probs": [0.11340415477752686, 0.6733114719390869, 0.06537741422653198, 0.14790689945220947], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "99e9e4de72d1f66f690cd6d56b59be900d22276d7d3e47b8229af9f1da8bc2df:action", "state_id": "d05ae54988a5bb72bbcdd85a90a720dfc8696bd500ad45d83b9dd104186ce7be", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.04296875, -3.5546875, -3.13671875, -1.95703125], "student_probs": [0.3780394196510315, 0.08336926251649857, 0.1266273409128189, 0.4119639992713928], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d635400e35a75d8a18c368731d02524cce67d3cde7fb4cf4447a2ab612f9f131:action", "state_id": "cf6f233f7cd312bb9c762eb650c3ea6e3b90705aa564d2cd2264003ee70f8aab", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8642578125, -0.810546875, 0.708984375, 0.03125], "student_probs": [0.10722692310810089, 0.11314365267753601, 0.5170758366584778, 0.2625536024570465], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "af42b16eb4519b462f1b2d5d8c600054e867aa1b16c49a283425308813aa9fdf:action", "state_id": "f040cdce15f756992046fb66eb3f080a2c8c019e7216f0c3f9d708a8ddc7af38", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.14453125, -0.1875, 1.064453125, 0.90234375], "student_probs": [0.12259775400161743, 0.11744145303964615, 0.410712331533432, 0.349248468875885], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "75cc116e37ba7ffa4f2604f08c4aac4edb9e52ecdbc31bbe5074ca60abbc9434:action", "state_id": "41327aefce319982d1c486b838970cabda2b35b67b1cf4ead2b5f80be9d26889", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.421875, -3.0, 0.01171875, -0.091796875], "student_probs": [0.24938993155956268, 0.01893273927271366, 0.38475677371025085, 0.34692054986953735], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4877cd956103c385773f1783e9fbc90d2474da652a176cb97a5e3564253c9d10:action", "state_id": "e029afc44caefbe3fc055bf5d7d5c4acb82d27437dd31e19981cfc1fda6da2a3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.38671875, 1.15625, 0.25390625, -0.4921875], "student_probs": [0.04690001904964447, 0.5964449048042297, 0.24192872643470764, 0.11472631245851517], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5e68443b8d027e25d3e1b0746cbd164a6386325629e2bf68579dedf7dd76a36d:action", "state_id": "bf1d841ba6d72c944b13ffad8e3046235db1fe5be09bd733f1d97c39a5e2eac5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.583984375, 0.638671875, 0.4921875, 0.3515625], "student_probs": [0.10123269259929657, 0.3438061475753784, 0.296958863735199, 0.25800231099128723], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cd725e4198d3e278f685331a2dc8a3418e10d5ee544c1f84e97fb51d78d72b3a:action", "state_id": "1c6e03c0bd0d90dc0e40c4f7da3d539bc82ca19977672137c452f585ff82fafb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.49609375, 1.44140625, 0.359375, 0.48828125], "student_probs": [0.07710105180740356, 0.5351873636245728, 0.18137842416763306, 0.20633311569690704], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9545b42ae566c8979faeda45667ab78e40e9ff44c4642c256f3c6f1caeffbfa0:action", "state_id": "e2d3d28029af2598224717363fa452644470a04a9f6ea6bc8f6458c76000e6a9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9765625, -3.34375, -0.337890625, -0.5244140625], "student_probs": [0.2193274348974228, 0.02056063339114189, 0.4153982102870941, 0.3447136878967285], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5dd51aff9369dc487b08ade82f62e58ab7e4cd2ccb744ee5ec2f300640fcd5b2:action", "state_id": "4c6df7cb50895e95ad66fee5fef7619a91f8f20f3909ea86e404c80b4077e906", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58203125, 1.21875, 0.15625, -0.4560546875], "student_probs": [0.03812677785754204, 0.6274714469909668, 0.21684834361076355, 0.11755348742008209], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0405edc82b5c28c7dc256335c6856b801e1b2afb9c028688ec454552b39c1e2d:action", "state_id": "825bcb3a8800673559e9abf5f883cbca7c9da4d3d0ccc9c6e9bf4c6405e9bf30", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.072265625, 0.7265625, 0.19140625, 0.390625], "student_probs": [0.06711690127849579, 0.4055580496788025, 0.23748578131198883, 0.2898392677307129], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9480f67f807181991bd86a3fe013bdb07bac78db8896d3ac832eced432e7c25b:action", "state_id": "6db4ae46d468814cb8a0f8ef3934eb40e77fc9002451b97a7fe15d4cc4fab12c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.228515625, 1.515625, -0.29296875, 0.23046875], "student_probs": [0.04273241385817528, 0.6645421981811523, 0.10890811681747437, 0.18381726741790771], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fd553614a165435159220c5e478e34800f70e8296124e65bf784c078012697dd:action", "state_id": "99f05452b734480f19b9146408849f2694f95efd50ccb92c626499ab5e4c2d55", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.43359375, 1.8515625, -0.80908203125, -0.4091796875], "student_probs": [0.030896663665771484, 0.8253474831581116, 0.057694390416145325, 0.0860615149140358], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dfb611d4a82de8e8b616feb90b0f982e2e4e99e4759266c543d710cd2581189a:action", "state_id": "a9edb56bc8c5390612b705182bc714849c9c65646e8480f812ea4070f4ccb00f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.47265625, 1.95703125, -0.587890625, -0.337890625], "student_probs": [0.02673806995153427, 0.8253239393234253, 0.06477075070142746, 0.08316728472709656], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a77d6fc4d0f80979434874afaea506fe68a9d2bce826d92c96a51521828ed36b:action", "state_id": "2aa7c80d48f4035edb1afeef6810fffea65801dd1da67d87d8b58260d668a572", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.41015625, 2.22265625, -0.4365234375, -0.08203125], "student_probs": [0.022104069590568542, 0.8359545469284058, 0.05852152034640312, 0.08341988176107407], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d10f7f7f7b34c62d2d06b480ec5b8fe9395d6deb8c25e0bb04abee0b3cd5e2c3:action", "state_id": "cd3884e8190299bf06f0c3c92779bf070e06660ae9f6e6f288d1e9c3fb52dd96", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.30078125, 1.931640625, -0.43359375, -0.015625], "student_probs": [0.030924905091524124, 0.7836666703224182, 0.0736076831817627, 0.11180073022842407], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5a3d33df75f8ebaa6d340b578b6cf79c7967ca0cecb53e8645db136464efb9c6:action", "state_id": "1bc4a265b8351dc7cb99c68502ee635e39e4bbddcac3aa96313993a15a3a2cbb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.130859375, 1.98046875, -0.4970703125, 0.1640625], "student_probs": [0.034499067813158035, 0.7745330929756165, 0.06502171605825424, 0.12594610452651978], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b195aaf784ad958abe48e771b17cd10f9accfcf39bb605ec525c5fdf8a9b4767:action", "state_id": "f82187d602b4c4136a6af6751b5aeb9da2990acb4a6fc9dbe056f51eaf309330", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.10546875, 1.693359375, -0.799560546875, -0.37890625], "student_probs": [0.0479588620364666, 0.7877429127693176, 0.06512131541967392, 0.09917699545621872], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "db8fbbe977566836685fde805bce9687cc9f7f609a8386626fcc670b40462b07:action", "state_id": "77e20beb3fec805ad6fcfa2f0a1a8ec4421e93b80e00e2e4b535af3df2e19d42", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.056640625, 1.82421875, -0.6376953125, -0.33984375], "student_probs": [0.0446472242474556, 0.7960416674613953, 0.06787972897291183, 0.0914314016699791], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bfa603917a825f80ddfa0e35ef92c4349ecf5950f7b810f4927da9d40fa4f63a:action", "state_id": "b11be583c9de7c392c722ac915ef988244ab43ecb842064dd2e45bd0c57401a3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.875, 1.923828125, -0.4482421875, -0.078125], "student_probs": [0.04722268134355545, 0.7756508588790894, 0.07235844433307648, 0.10476810485124588], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4a3dd60d38bed18f884a308722156e0b45c2e52696e6167c80d47d13919a2d00:action", "state_id": "39f88e4bb599c444cc712f1bac3a4cb8e8657bbe098a239ef21e96fb45250194", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0546875, 1.62109375, -0.390625, -0.162109375], "student_probs": [0.05023162439465523, 0.7295486927032471, 0.09758339077234268, 0.12263628840446472], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "829fb2bd09472b2ac89fb47c41bbde1c6eed2a3a2ddcf1fa19dedd5345961080:action", "state_id": "32a82ebda037b8aa9f80cc72b007d1bee581f03e3eeb75c10e69a4d71cffa2ef", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, 1.8125, -0.201171875, -0.015625], "student_probs": [0.04263873025774956, 0.7397251129150391, 0.09875151515007019, 0.11888464540243149], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5acb772c6ad5eb40d25b9a2136cfbf257fd163c032c42ce20b076e27c47e36a0:action", "state_id": "ad00810f86aeb01e7aac6e6312ce61668141b4ea0456d04ff8552067d1932c3d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9462890625, 1.890625, 0.30078125, 0.3125], "student_probs": [0.039897359907627106, 0.6807698011398315, 0.13884809613227844, 0.1404847949743271], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7e1532471bfc53d043ba3abb3fa48ae9ca58ebb874237e52ad2835346abe1cb9:action", "state_id": "ae6c2c751366bed2d5658c14d694a4a7b129889a205539c2c11c1d7d7cb3381d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.65771484375, 1.96875, 0.568359375, 0.58984375], "student_probs": [0.046052251011133194, 0.6366636157035828, 0.15693800151348114, 0.16034618020057678], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d30f67af1fb6d50b9c92c14a03def4a5bdfe1e968e33f3f97246996bbbd26800:action", "state_id": "23e6378346cbad16aba594541b85a51da1be8db6cee0fb799acf0848482075fa", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.078125, 0.7890625, 1.431640625, 1.298828125], "student_probs": [0.09712057560682297, 0.19772768020629883, 0.3759547770023346, 0.3291969895362854], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d8f9189b0cbeee0369e5fb2c2b9dd122281981b5b62683887cc5a6decb377eca:action", "state_id": "408a075ba78cefff5a24910293795c2b8059545113e0a8243a02fd73d4d04ab6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.73828125, -3.1796875, -0.13671875, -0.19140625], "student_probs": [0.21552415192127228, 0.018758870661258698, 0.39332470297813416, 0.37239235639572144], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fe0bf8518842102c20a2b77b35ccc9125503376a5579b6f4fac1c2df088957a4:action", "state_id": "f6f95152005836a0e82b143f5afd5a78fd8ea50ba3db86727c6ebe1ace2db7a0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.65234375, 1.017578125, -0.1015625, -0.6787109375], "student_probs": [0.04385669156908989, 0.63323974609375, 0.2067909687757492, 0.11611255258321762], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "44cd596dce8b824d762f77c71e252319cff825021fbdb6a6a91dde723b6ec8c6:action", "state_id": "02fa4a7646fbb0f624a5ede2e7e76c2500516b4c473b5367a7031cfdcdc353bb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.31640625, 0.1953125, 1.5625, 1.32421875], "student_probs": [0.12342192977666855, 0.10934577137231827, 0.42910540103912354, 0.33812692761421204], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "684d2e73d65a064cdc5c94db9dce9a5baa19d1173264aeaa7f50aa14219db440:action", "state_id": "e246b63cda1e1333c46d40b5012989e1c831c8e1ee670820cd5016e6e513249d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.46826171875, -3.12890625, 0.0546875, -0.0703125], "student_probs": [0.23553423583507538, 0.016464585438370705, 0.39734524488449097, 0.3506559431552887], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bb8b92b9d1da5ad5432baa864995f2d6629cf3d4dbbb78db403e4b13ef92629a:action", "state_id": "798c4d2f09b1e4eaa0f1f78df267fab7262b3a83ade39944d4c0cba5c114af2a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.61328125, -0.826171875, 0.3125, -0.888671875], "student_probs": [0.08249825984239578, 0.18125168979167938, 0.5659798383712769, 0.1702701896429062], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a96fecbaf5a9031ccd28b01249e1508348faeb8afa4084a9c34a956a3950adc5:action", "state_id": "5028e53dde86cb00c50ace6b0a903a12467aaabadc4d9c8d24c68338c598622e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9912109375, 0.296875, 0.09765625, 0.1875], "student_probs": [0.09219199419021606, 0.33427339792251587, 0.273893803358078, 0.2996407449245453], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8cb5d85921f1bc7e8e6808215ac7e04ba5ae763c000369f04ab0d3bd441db549:action", "state_id": "79b0ee4ce83c48f6eb3c250b2966b37e21777dc0c69ab2339c498d9e3b308758", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.75, -2.8984375, -0.0703125, -0.134765625], "student_probs": [0.2024284303188324, 0.023616576567292213, 0.3994441330432892, 0.37451085448265076], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "56e7504722cd661bcabbf3297407ebc92cfba1d98d8305eb06e9f34c8be88d25:action", "state_id": "275caeb47aede9a4ad5afd274f9e5b4eac9afd054100149083a40b07cd53d64b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.935546875, 0.96875, 0.1171875, -0.46484375], "student_probs": [0.08209317177534103, 0.5512297749519348, 0.23523598909378052, 0.13144098222255707], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "32fd90e1da232f2070d689bfe014a908a07bfd13f99897b4e73cc025e6951a1e:action", "state_id": "69ccd71e200cc26242ce6d9f1392f6d9cfa9d7e7fa98fc996b8e4e95bb694ebc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6416015625, 0.4765625, 0.4453125, -0.05078125], "student_probs": [0.11325272172689438, 0.34646639227867126, 0.3358067274093628, 0.20447424054145813], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "52c797270da6d0a9e41e1292e9eaf273b8214b5260ad80e2e7ecbd6e6724a9cb:action", "state_id": "c7817cb1cdc0b65bb609d7e3113246f3e78d0083e723bfbf0d89e92ddad1ebd0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5537109375, 0.732421875, 0.3984375, -0.015625], "student_probs": [0.11207292973995209, 0.40556561946868896, 0.2904113233089447, 0.19195017218589783], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "393cfc35d71e369ed389cfd9b0abb4c85026c7c5a9dcde6a56d09b6a68ae4361:action", "state_id": "712a3efe67aec893c61b8b49c50f3fbfa889108724739bffcf2059bd16c19f6b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6328125, 1.146484375, 0.21875, -0.08984375], "student_probs": [0.09099096804857254, 0.5391840934753418, 0.21321961283683777, 0.15660534799098969], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "84a61a95c9e465056597546866be820cf6ec0bcc6608822ab2827c688e5e5793:action", "state_id": "70079b2d357ebe3f1db085dc6fd1d5c73b307f078137240a27b4538afec687c8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -2.65625, -0.224609375, -0.6541748046875], "student_probs": [0.20301082730293274, 0.04028873145580292, 0.4583863615989685, 0.2983141243457794], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1493fa61a381c97a17da870a7271a4e9a8cbdfd196ddca0ba00cd316f43f1e2d:action", "state_id": "7b98a8997e3085c8508e16b258bb3b6c4a145ebb80b1a406ebbbc389ffcf38db", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.14453125, 0.1796875, -3.01171875, -2.91796875], "student_probs": [0.032078418880701065, 0.8910515904426575, 0.03663470596075058, 0.040235355496406555], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "68a06f8ecfe4f953705a9fa655ab48f502429c7301b1a14a47f5a6e2d5f9d4e6:action", "state_id": "50e564d14adf49d113e853c1a3c2656fc638d874251f82edb7978662e162b4f5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6953125, -0.03515625, -2.3671875, -1.19140625], "student_probs": [0.118679478764534, 0.6242697834968567, 0.06061554700136185, 0.19643519818782806], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "78cd4295cf85566bf6302778a95d194322fe8d94f9676c9f8d5f252b108efc83:action", "state_id": "01365ad83ede9c7f8d9d4bbcdc73c48dbb0ec0942bec0485fd2793262fcaed9c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.98046875, -2.5703125, -2.19921875, -1.6796875], "student_probs": [0.26962533593177795, 0.1494840383529663, 0.21665003895759583, 0.3642405867576599], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "facb01ad29bd6e29bd21e892c2f1174c7fce6bc7d9291560f09faaac31fea322:action", "state_id": "04af46adb65a85dac95b389c9359fca5296c9ed6a84d1d3503ed1654947b3004", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.671875, 1.216796875, 0.04296875, -0.490234375], "student_probs": [0.03599070385098457, 0.6467323303222656, 0.19995741546154022, 0.11731953173875809], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "017adc636463905e5d32aaf5fb318a3767f2cf6dfa16ce06be2eb06eacb6ad93:action", "state_id": "76dc25c0269c8e7750b5c413128114b38cd7526ad9e3a240ea82e094a20d9bde", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.817626953125, 0.5859375, 0.55078125, 0.5234375], "student_probs": [0.07799167186021805, 0.3174011707305908, 0.3064364194869995, 0.298170804977417], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ab2a9dd55da707ec8f757a7209a65aa975a0846bed1cacb4ca36ec7815c3668c:action", "state_id": "45e4feea189636d3c52f3571adcefb5565da078e98792e249687654bd38920d9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5966796875, 1.373046875, 0.56640625, 0.693359375], "student_probs": [0.06666028499603271, 0.47786861658096313, 0.21329905092716217, 0.2421720176935196], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d838f2eeb10cf728e510be2bfbed5e5924fbe71bf0c56ca2fa9055397543470b:action", "state_id": "45760c98c3491e1193a782a8f2f54eac60f51032d29cef101f6dfd66801254f5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0859375, -3.31640625, -0.32421875, -0.48046875], "student_probs": [0.19679060578346252, 0.02115066722035408, 0.4215165376663208, 0.3605422079563141], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "03ade29e8a385246436aeaaf77fde4dbe9431564f48ce33d956608cfc524f82f:action", "state_id": "84fc49b8d4a0f33f5d3155bfdff6712f7e56aa39ea8431b09f89fe1f915d6096", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5078125, -0.70953369140625, 0.53125, -0.4599609375], "student_probs": [0.17565728724002838, 0.14356869459152222, 0.4965068995952606, 0.18426711857318878], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f6e7ed79eb24f0325cf77de7202328987d2ec70ed59f0daee55f86ba855f78ab:action", "state_id": "d6d9b02b7d566a59185aef7fd4d702a972fa5ec823b5d25b7246e98d56fe3096", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.04296875, 0.0078125, -2.22265625, -1.27734375], "student_probs": [0.08503516018390656, 0.6610609889030457, 0.07104954123497009, 0.18285433948040009], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e129fe338223adbb04ffd9696f6413fa3bf6c25aff40823154784c57c800bfb1:action", "state_id": "9ae76a84006e5e02f65dc25f7bdb0ed12ec00c36afbd7d5e38b58f01e14f2507", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.93359375, -0.0390625, -2.52734375, -1.2265625], "student_probs": [0.09775509685277939, 0.6500157713890076, 0.05398549512028694, 0.1982436180114746], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e61d386168d3c25d8c515f5eeb900b9f95708a13be9aded2ffc245aa77b5aed9:action", "state_id": "0d6ccf2f1d99e35e00ab57008e23b7fadd779b6b548b3b4f6911a59c1d82bbcb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.921875, -1.77734375, -2.26953125, -1.578125], "student_probs": [0.2340788096189499, 0.2704775333404541, 0.1653396189212799, 0.3301040232181549], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8140cc0bf909cd1bd1f37094938ba3c7b3017b09de6ce7ca732480cd79aa5af8:action", "state_id": "5e22aa8ad406c3844448b06cef1d979f60e6b248da5a7873bc19b866e70a9e55", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.935546875, 0.96875, 0.1171875, -0.46484375], "student_probs": [0.08209317177534103, 0.5512297749519348, 0.23523598909378052, 0.13144098222255707], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0c1bc2c0fb590848379cc9ce2069d6b3985293cfe16271cecaf188629e4f6c7c:action", "state_id": "f0662d253ab524cb06f74c94f23bf89d537bd3c7fe29a26f040d6d671f12f745", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6416015625, 0.4765625, 0.4453125, -0.05078125], "student_probs": [0.11325272172689438, 0.34646639227867126, 0.3358067274093628, 0.20447424054145813], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b045c4d266112dfbca2a72f4c8439eef82ca819de8b4cd86b26a5e177bed43e2:action", "state_id": "a3f2a5e0b0a7bc68c7a6188f72a828d527983a9c9d4cfac84a79b9ada0f1d093", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5537109375, 0.732421875, 0.3984375, -0.015625], "student_probs": [0.11207292973995209, 0.40556561946868896, 0.2904113233089447, 0.19195017218589783], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b8a14df57f22328565d8e380857e22dd13742e3b3d29136a69ddc14ebaef58a0:action", "state_id": "7ccfdd9fdaafea12440e97c090eb3e128924f6810d43a57e15ebb350d27c839f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6328125, 1.146484375, 0.21875, -0.08984375], "student_probs": [0.09099096804857254, 0.5391840934753418, 0.21321961283683777, 0.15660534799098969], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "58064dc6141027162daf63878e92edbc4c0a874a8ade899ef4b0fbeeab47ef2a:action", "state_id": "e42d1257b39c73271ca37ccec9372524925d32bd92c1b8e71a8292b10bf0cc7a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -2.65625, -0.224609375, -0.6541748046875], "student_probs": [0.20301082730293274, 0.04028873145580292, 0.4583863615989685, 0.2983141243457794], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "deb8bcb34d581ca763dceadf61d4f029fa31435c72393f92a239afaed561c16f:action", "state_id": "4783d303e57f996bf89d899c99a34eb17812cb1bfaeafb2bc7f3753c0234ea16", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.40625, 0.2578125, -3.0, -2.9765625], "student_probs": [0.023224761709570885, 0.906219482421875, 0.03486450016498566, 0.0356912836432457], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dc6adc76f0b2975a861ea9a41699dff94331afd09a69bc788a2daf78347ea096:action", "state_id": "30d5b80cc0842154323bee11b0d40415c03bec3eae35e237ff97a5532d4d6217", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6328125, 0.25390625, -2.20703125, -1.3671875], "student_probs": [0.10565156489610672, 0.6970556974411011, 0.05949711427092552, 0.1377956122159958], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "74ba5526af738feebf473b0a247209cea495a1f0d3bb14e6017df0c69095d3b6:action", "state_id": "7cf7515767676f7c2717ed1024a243204e76e7ee96fa4aa97d9a80c3334b046c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.73828125, -3.53125, -3.5234375, -2.4296875], "student_probs": [0.3058050572872162, 0.1383766233921051, 0.1394619196653366, 0.4163563847541809], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "64f12a54cc0b55ad0b2faea010f596954ded2ed2661191316b107d2b21ad5d01:action", "state_id": "b5d1f1eed2f45879bfeef5d9985d5d3cf2b272f09fd108acd85711cc2add99dc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.45703125, 1.04296875, -2.59765625, -3.046875], "student_probs": [0.010538976639509201, 0.9486884474754333, 0.024889735504984856, 0.01588279940187931], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3cd4b3e2dae183f44b453f86bcfeae481084db65061bfb7d4f64af84df5bb922:action", "state_id": "fe636748f86372e7312573933dee13291dbbf26db4b14c0e527c49411dc851f8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.41796875, 0.72265625, -2.390625, -1.80859375], "student_probs": [0.037057191133499146, 0.8566997051239014, 0.03808445855975151, 0.0681586042046547], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "72313f3ee7e4b8ec18d12071e014b6e338703a7e7474e106d70a6b23e2ec29e1:action", "state_id": "434f8698712a5155a4ee96071e8ea628d8a8660a86ee3a347b82de941b3f8174", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.046875, 0.99609375, -2.37109375, -1.2734375], "student_probs": [0.040228988975286484, 0.8434972763061523, 0.029089264571666718, 0.08718440681695938], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7edb9adb3e7dbc67e410ac5e3280473fc03c7d4b5474f09dfb3cf1b88cf376ee:action", "state_id": "960e4897edc7b61abee51f2e72e2c1821ad2e804ccf2ac1a54fda31f2e10428b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.00390625, 0.55859375, -2.55859375, -1.578125], "student_probs": [0.06221523508429527, 0.8068194389343262, 0.03572720289230347, 0.09523820132017136], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "27c81317f251ea6acefcadd9f5c1f469854dcb8ef5382eba13f4ca3afba505f7:action", "state_id": "be62e9f03d1e039c67d06b86db74b3461594be8bab13de4226a1da58a5a60e84", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.73828125, 0.9609375, -2.484375, -1.69921875], "student_probs": [0.0575302429497242, 0.855366051197052, 0.02728172205388546, 0.05982198938727379], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "78ce9bd61ec5257691c7b9077b2dbeff12addffd20f47f5a1979a4645437a3de:action", "state_id": "2eacff84dd4fbaa2fcddf0025a1c2a3f52981ca65f18cb1911912a700074ac09", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8125, 0.8203125, -2.62890625, -1.546875], "student_probs": [0.060027219355106354, 0.8351494073867798, 0.02653307095170021, 0.07829025387763977], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "030d1e0cc74e29846e266e5298c64efc3a95a3b123762ad5f0752c592d07f8d5:action", "state_id": "7c23458b4c7d4e3a06d17b4746bfa9902ae4fb1de75b33e9b1f60ab3e23e4def", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.08203125, -0.07421875, -2.53515625, -1.7421875], "student_probs": [0.09535273164510727, 0.7100926637649536, 0.060609884560108185, 0.13394466042518616], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6d21f46bed14cdc12ff4a45871239e388aa8c06bb602393bf93f68afbf34f2ab:action", "state_id": "cceaa237cfb75344c50b91b9b9c6ce7c43c0e3fa9b13a2a65143762e0072aa0a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.72265625, -2.0234375, -1.76953125, -1.5625], "student_probs": [0.2585221529006958, 0.19136837124824524, 0.24668356776237488, 0.30342596769332886], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9b4a3c48fb020e7d5e5a83630f7c3819ff9ffd671b8fe24a42f85c50952bdb61:action", "state_id": "473dfa684d3700ba05fa9609f30d2ea7ce380bdf40a15747fc21ca661c26e8df", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.8203125, 1.03515625, -2.04296875, -2.11328125], "student_probs": [0.019064173102378845, 0.9007967114448547, 0.04147764667868614, 0.0386614166200161], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f64964292506205bfe4a5712cbdda047a9158aa52ede5e58d1214c614a356d70:action", "state_id": "ac064a1aac1b588290a5c30e9ff4a7d42d9af63602c0c7b6a68fcf28e76ced7d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.92578125, 0.65234375, -1.859375, -0.8447265625], "student_probs": [0.05497869476675987, 0.7242022752761841, 0.05875357240438461, 0.16206547617912292], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f0943b3573b47bb443a9731b9993ae23cba7bd1c8acb9ffa933fcabfffb0c67d:action", "state_id": "134374613915a8814b074f79a53e61ccc401549671e76a369bdaab9d5d87b77f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.109375, -3.33984375, -3.7109375, -2.65625], "student_probs": [0.25540587306022644, 0.20283344388008118, 0.13995087146759033, 0.40180984139442444], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e43b2fa4fe40aa105bf66dbee5d9d5f1ec1880a6c7197a4d22f478dc60c3347e:action", "state_id": "cc66b93153e93ac79d08f8745a8aede140a75f345e145aee693f6c2356b60abb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.56640625, 1.16796875, -1.5546875, -2.05078125], "student_probs": [0.02114752121269703, 0.8852744102478027, 0.05816253647208214, 0.03541542962193489], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8b0b4c10dc0f9a4f1e5de09f0fa0c82d26231c1663b5554f2dd9b39cc79b0db2:action", "state_id": "4eae961f0c480b07c4e0269c91774c845ab4e25332a8af342fd0ae5acfddc6e5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.33203125, 0.76171875, -2.0234375, -1.22265625], "student_probs": [0.03642507269978523, 0.8035241961479187, 0.049593064934015274, 0.11045766621828079], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "adba29466b4030b028e5e2888af1435157e6487d1248b53995821b32e45fec0c:action", "state_id": "8a49cbfdf82f738b67af27c0da359583caf81fc65a6376686008ca811ce63b37", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0546875, 1.20703125, -2.03125, -0.955078125], "student_probs": [0.0321325846016407, 0.8384788632392883, 0.03289458900690079, 0.0964939221739769], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "674608a77770a75ec151d766953ae5749168ccaf6c070ed725b93213841a6795:action", "state_id": "b5541c89668eb1e59794ff61c9cc5651f2c5e68462da7948fd772e761e3828ad", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0859375, 1.21484375, -2.19140625, -1.19140625], "student_probs": [0.03176628053188324, 0.8619408011436462, 0.028586555272340775, 0.07770632207393646], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f7dac5d271a6a39ddda34e4e22fff5c93637c8117e23565f7f5c8c4809f0e169:action", "state_id": "77e50be02a7466824281381dcce3f28390f03c83307c034a85ee11f363a1de69", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.8984375, -3.109375, -3.125, -2.52734375], "student_probs": [0.2465232014656067, 0.19964058697223663, 0.19654543697834015, 0.35729074478149414], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c3f570f8c80d9dfbde092a6fbd4f42fd8dfea14a3c269adb937d6697fc15c8ee:action", "state_id": "7c5321ba19be1eba2b50e357b7a751119eb44b05fdb34bd5e5d6d06ed0693e4f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.04296875, -0.56396484375, 0.994140625, 0.30859375], "student_probs": [0.17134243249893188, 0.10176517069339752, 0.48336565494537354, 0.24352669715881348], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5895e42cbeaba3c80627fc1f5584781356c0f95191defd11e0a4b7d9c5ffb5bc:action", "state_id": "72604c26db138b28039338dfaefb5e652430153b852863a71cf0910656b972f7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.111328125, 0.27734375, 1.75, 1.552734375], "student_probs": [0.20478494465351105, 0.08894123882055283, 0.38785526156425476, 0.3184185326099396], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2bf18a374e4428e6951b5017ddaa29022809595f0c97aabf0b7538cad2bda638:action", "state_id": "da781dfa208724da186f55512785e774c0ff5a25676e6fe852b2111ea95a7a7f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.46875, 1.421875, 1.2890625, 1.392578125], "student_probs": [0.11927585303783417, 0.3093780279159546, 0.2709004580974579, 0.30044570565223694], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "039d95695639416ed952e770eeab203d65d370e5a040b298b54c8d52d4eef55f:action", "state_id": "902ff77e7663cf625e28f4bc6fb95262c11cf54713f5c0e8163e73fca41abbd0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3701171875, -3.03125, -0.0390625, -0.103515625], "student_probs": [0.2654050886631012, 0.01854359544813633, 0.3695595860481262, 0.3464916944503784], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bee85ef1ffc88627c52c3e7368cd9e6678faf1d21710e2ed5343ae56d2e8523c:action", "state_id": "355ae9b3adf0df2bd02d4172aaa1238a696fef43284d5649ccda5dd36dceee88", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3046875, 1.1953125, 0.31640625, -0.439453125], "student_probs": [0.04850442335009575, 0.5909048318862915, 0.2453654557466507, 0.11522530019283295], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "74ed307e53d1da0b0aa0aacbd6db9c26d97556e2a3f3df0ebb66dccab627d54d:action", "state_id": "0bf2adff9d103f4796a3a9eea62dffa52da640fc389177c6b7fbf0ab11491e35", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5830078125, 0.595703125, 0.47265625, 0.3515625], "student_probs": [0.10341064631938934, 0.3361034095287323, 0.2971900999546051, 0.26329582929611206], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ceb3d236e1f8f94b510ccd9b4d314d5ec488cf5733499132e3c67774cfa572eb:action", "state_id": "585abf9afcce683942d6ef86a817ad92df0adb958a96b1f739d6c9946a4ef88f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.076171875, 1.373046875, 0.109375, 0.41015625], "student_probs": [0.04932764917612076, 0.5711795091629028, 0.16142356395721436, 0.21806930005550385], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "99558c018d4311e55eec34ef89d3b127973fd22294a780806449045ec1a5b0a4:action", "state_id": "db80e19fc8b5329bb3e88988aba6e511ead727e2f49e5cfffc8b4b90e13564b7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.244140625, 1.73828125, -0.2421875, -0.01953125], "student_probs": [0.0372273288667202, 0.7347020506858826, 0.10139220952987671, 0.12667851150035858], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3b507d093cf02cea73eeb03c2eb8e2902d815ee92bb05f850cc9d24979c0187d:action", "state_id": "4ae1f3e58b2a6ce7be8e92b074a4af1c1cc584339dea1ee68fc2c8f27af8b643", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.52734375, 1.8671875, -0.713134765625, -0.4619140625], "student_probs": [0.027808552607893944, 0.8287138342857361, 0.06277473270893097, 0.08070280402898788], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6dc7e9bfd8804fa3043e52fc252b6a0e7598613887c1577c6cf3bdf339d43b86:action", "state_id": "a32bae3124a04d889630289b4a53562cc6ab1469b8e302eb52a55818c6d97f0f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.48828125, 2.1796875, -0.5517578125, -0.177734375], "student_probs": [0.02153707668185234, 0.8436558842658997, 0.05494316667318344, 0.07986380904912949], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "70a4b705ae4ecd52260fbca6bcfa2a03513e2e390ae3209bc7f9b5ee0f428673:action", "state_id": "9bb7f0bce3bf476ed6dac52fc14bbc35366f1d2aaa43dd8672fb57792563a8ac", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4296875, 1.8984375, -0.5732421875, -0.140625], "student_probs": [0.028677813708782196, 0.799709677696228, 0.06752980500459671, 0.10408275574445724], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e61ca4ab6c9020a362cd4b61980c902fe5fe7806f6b9f6ed4badc387f60288ef:action", "state_id": "22d023a8e89aa4c1254a2d30f8b488e672cb69a2975a27425f3e22926ecd1b6b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3359375, 1.958984375, -0.6822509765625, -0.1953125], "student_probs": [0.030278582125902176, 0.8167740106582642, 0.05821407213807106, 0.09473329782485962], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c2a964f93848a02f5307ae5f0dd18df0ec97913452ed16af82da9c586eb78e13:action", "state_id": "85cc749e21d5124760510e21cd0cb0a50c92c68703732efc5e55d255aa4f4e84", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.158203125, 1.6796875, -0.791015625, -0.439453125], "student_probs": [0.04634943604469299, 0.7916344404220581, 0.06691322475671768, 0.09510286152362823], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1c3297603e6b1ad500ed2baeb8bb33f4693c1a19eb1066b87cdca16ff24b75a1:action", "state_id": "dd4dea24c2e43df30a2d7caf9773c78817cd22d2c3b0594ace1a4c740c671212", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.19140625, 1.765625, -0.7171859741210938, -0.369140625], "student_probs": [0.04145390912890434, 0.7976049780845642, 0.06660652160644531, 0.09433458000421524], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7cd9a927516a94a8f23718310ac87f4d3ab274069d7d75f3bd53b64ea5e73ad7:action", "state_id": "a50f6973eeeb1ab6ead4651a1e268bc5d7b214a5ff596ed810653c413f5346f6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.171875, 1.904296875, -0.5791015625, -0.216796875], "student_probs": [0.03692333772778511, 0.8003232479095459, 0.0667942687869072, 0.09595908969640732], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "da56513ac3521af105f734fc6bc4c828cd2bc9117422b41f211cc25b349585b6:action", "state_id": "eb229af872e2f71d603d7c14bdb82cecb035718714398ffd6a1dd2241c4e2835", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.234375, 1.64453125, -0.369140625, -0.119140625], "student_probs": [0.04128709062933922, 0.7346954941749573, 0.09808006882667542, 0.1259373128414154], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b9a6ba3b62e900db6ff0c650f356b526244c83571a5aa5caeed63bc79e72f5de:action", "state_id": "697af8221699ce64d630f4694cda0fadfd74b24833342aa6f933a75fb9cb8793", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9853515625, 1.912109375, 0.03125, 0.17578125], "student_probs": [0.03986383229494095, 0.7226539254188538, 0.11017511039972305, 0.12730710208415985], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8120b16a21d9aca803bc8d7b2279cff22a1659b3e4460eb5f90ee842ee244958:action", "state_id": "2d75caca5712df0de72ebb8ef876cb3538b8456b50a3fa4a708ef1d703d8c70d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9619140625, 1.939453125, 0.4140625, 0.4609375], "student_probs": [0.03662079572677612, 0.666462242603302, 0.1449795663356781, 0.15193729102611542], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3fb578613adca74bf72759a926bde300625d539f4eea46364d2e6d43e1a6741f:action", "state_id": "6b05735a01007e944a28bbac572b85cc29283ca7bd2738b96beef6bb3a19061d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.90234375, 1.88671875, 0.421875, 0.46875], "student_probs": [0.040056608617305756, 0.6515513062477112, 0.1505827009677887, 0.157809317111969], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "88e2de71b0c4aece815de4cc3b53d339a26a45ba25e8217eb5f12ce27ab5d8f4:action", "state_id": "671eb9d62b0a16f25b9bc909e3c7f9725be9c018b77afdf5936bdcfbf05a90a5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.05859375, 0.818359375, 1.40234375, 1.40234375], "student_probs": [0.0925535261631012, 0.19785861670970917, 0.354793906211853, 0.354793906211853], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e02ee1dc5145d5937baaed94827a6230b0e83e2b97df10fdc0636a487d95d418:action", "state_id": "4e50fb3deec687be43174eab65ff59026e1d68ca4cf03c2e7bbc7126d831f228", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.530029296875, -2.6953125, 0.125, 0.1171875], "student_probs": [0.20201475918293, 0.023174617439508438, 0.38891860842704773, 0.38589203357696533], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0b6abd4707e372be4bac566d3bc6ae4a6f6143c103e9bf850c010c98edeccaa8:action", "state_id": "e1652be90b91dcc87ddebcbf44a60205b035aaabfe5d5fab4f8a7e099b871e48", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.484375, 1.232421875, 0.22265625, -0.361328125], "student_probs": [0.040455445647239685, 0.612162709236145, 0.22301355004310608, 0.12436839938163757], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6c8244f04b01af2c42403da2b571e0bc66fb67201c492414b616d58cb2b5b3f3:action", "state_id": "afdfdb266c275689e6dc6a44842bd576d8622d42cc51c23ca4a5d3b2e26276c7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01171875, 0.734375, 0.2265625, 0.40625], "student_probs": [0.06987840682268143, 0.40055474638938904, 0.2410580813884735, 0.2885087728500366], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "366c843fc436e636ba0dcbfdbf22b185493d84dafab6748318aa4145cc55bdb9:action", "state_id": "f449f3b4e3680cb52ec2718c17d54133d1beaf85572edfcf3da907eb880d8cb1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.115234375, 1.51171875, -0.13671875, 0.33203125], "student_probs": [0.04599066823720932, 0.6361228227615356, 0.12235836684703827, 0.19552810490131378], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1803ca683173e241fc51900f2cf9b9fc30144059e4e2a7c8d30f524a254e86d6:action", "state_id": "9a0365f1ee5efe734aeca6467cd0c2bb1c328bab3313ae304012a456cdde6412", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4140625, 1.9140625, -0.631591796875, -0.21484375], "student_probs": [0.029077885672450066, 0.8108660578727722, 0.06358951330184937, 0.09646657854318619], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2041b5676f55f6b16bbc9f091f899cfd855b7540295a3ac6557f1eecb5b42d62:action", "state_id": "107e2d0b50455379b1a7f7b0c5369a6fbafbe13563887a431d894a7115ab9862", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6640625, 1.576171875, -0.8251953125, -1.091796875], "student_probs": [0.03265228122472763, 0.833929717540741, 0.07554903626441956, 0.05786891654133797], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "13f3b59664320c0f60c546d88497ce1eee94bd278c675ca94e36e58d7701a969:action", "state_id": "346af45799f307aa630bd67bb32fdd095894140849b03ac300fd7f4e7a14867a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7578125, 1.791015625, -0.587890625, -1.037109375], "student_probs": [0.02436043880879879, 0.8470744490623474, 0.07848302274942398, 0.050082094967365265], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c6e4be584939033ff52fec3c51066b6d4da60bf150aa0a6b1bfd278c92caca97:action", "state_id": "79423b819f6cb52f5740451cbb7fd7dfa878a62f17cb28703d5b91b7e1d727ef", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.05078125, 1.64453125, -0.294921875, -0.5], "student_probs": [0.0508280023932457, 0.7527701258659363, 0.10823521018028259, 0.08816663920879364], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1b80ec093138d2cd0c2e9fe46f46f0078d577bb8e7649e8adf5c8628513efcbe:action", "state_id": "ddcffe9580fae74b88d2a5eb4e2bfb86e64988ec20c7f9f5cc31eda6fd80c6f2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.85498046875, 1.806640625, -0.91015625, -0.259765625], "student_probs": [0.055312108248472214, 0.7920408844947815, 0.052342891693115234, 0.10030411928892136], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "999cf7784d41b7df2a395712e2f7a9f224071eb172c8ba6de2d5ffaad3a0c0f1:action", "state_id": "546939730769944c2768c99645aff767ce8b0e7a87bbb37fada6c919b48cfefe", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2265625, 1.904296875, -0.9873046875, -0.734375], "student_probs": [0.03731365501880646, 0.8542456030845642, 0.04739975929260254, 0.06104106828570366], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7f8ba02deb814165216afff523a54335eb7750f04bd1f5ce051e21f9112bde8e:action", "state_id": "b5f4f124b5acd41a0589a02cb406acacf21cce0ae3bf7e3e09b685d8013b2319", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.146484375, 1.884765625, -0.828125, -0.34375], "student_probs": [0.03947946056723595, 0.8181376457214355, 0.0542791411280632, 0.08810373395681381], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ae789fa070dafb84a86cb406fe50afe43d8677974ebbc746ef5a753ee614fc7b:action", "state_id": "2e4e8f0a18a5440bffba0800b4906168003135752bc58e1cc22beea2eb25747f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1015625, 1.85546875, -0.660400390625, -0.177734375], "student_probs": [0.041128240525722504, 0.7913388609886169, 0.06393437832593918, 0.10359852761030197], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bb267ebf2f09cb64e7a4aa57df7121214566adb6bb431136fd047f8896c73afb:action", "state_id": "23fea0cd9c235f4b90a5aa0c5c6d5c59411a4960c66fe8dff85088d4203edca4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0029296875, 1.69140625, -0.400390625, -0.09765625], "student_probs": [0.049763500690460205, 0.7362853288650513, 0.09090553224086761, 0.12304561585187912], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "24eff71082cc9ec7df36d7ff5ba8b8266c82633c6e5b55a03a218c719950c30a:action", "state_id": "beec5c285e15f51b4e0194cf270e775dbb6d2dd5dd30e4b9a01ffd9a1b10b83f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.955078125, 1.982421875, 0.1015625, 0.1796875], "student_probs": [0.03867611661553383, 0.7297647595405579, 0.11125922203063965, 0.12029991298913956], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "83b765ab9a31793803bfaf09f545591b51091cf0669963d6c55ac6eabd4e44a4:action", "state_id": "c4ce4246f1ad52e97419cd5295c93904700d2840ce8fa180a546592df1f286a6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9130859375, 1.9921875, 0.48828125, 0.50390625], "student_probs": [0.036422330886125565, 0.6654447317123413, 0.14790190756320953, 0.1502310335636139], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9c2135f0eaa25d37e44949a54fc39a27ca4ee08af7bcaf052257449b4192d488:action", "state_id": "c90fe247a5c317bc5ed7e6f23970ce9b04acfbb0aadf4565867ddbb94ff3c93e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.74365234375, 1.904296875, 0.52734375, 0.47265625], "student_probs": [0.04532238468527794, 0.6401805877685547, 0.16154716908931732, 0.15294979512691498], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e4ccfe1e583cc73bb25785bcbe70267a8461d38c9198b04f8f41e30e2be5cae6:action", "state_id": "4135b01a57cfed2b4c20f91cd60ddf640d90775704b04520f17995c303a0222e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.150390625, 2.1064453125, 0.9609375, 1.111328125], "student_probs": [0.05840202420949936, 0.5579037666320801, 0.17744819819927216, 0.20624592900276184], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cbf33c9ecd20d741277e6b73e77bbf3ba8a1b9964140b1dd41fee137bb71ccaf:action", "state_id": "bf9dfc6c7dd93307f2367a7a95106c5804a3f965f445453491b739e85a89509b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.806640625, -3.16015625, -0.216796875, -0.13671875], "student_probs": [0.20606550574302673, 0.01958332769572735, 0.37168172001838684, 0.4026694595813751], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "56861ea5fcb90fb52bf26d7f893226df7c0a017a222d94ca849d46dd4f4e8b54:action", "state_id": "17aaf7c9976662e66cbe2f504c78db7ab62cd58afb675b5d854c839f4b56242a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.8544921875, 0.615234375, -0.07421875], "student_probs": [0.10066941380500793, 0.11943119764328003, 0.5192923545837402, 0.2606070935726166], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "55838a1b9fcad9f7aa3fee26e74af37b721716a749fc37b91ba42f88d49ffe4c:action", "state_id": "4e0b16422d68c5c1bfc7022368452f5ab8c0e0cf963d1e5a2409aecb1a832b13", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.21484375, -0.458984375, 1.060546875, 0.9375], "student_probs": [0.16951259970664978, 0.08640963584184647, 0.3948991596698761, 0.34917861223220825], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "56f363de55da02cca60fe2d99d74ec4211439e24a1a2e17fab73ef6e6b168047:action", "state_id": "6817723d2cb9e9c6025cd7fdf54f171cabe484e3baa8ba2ab92ae2a10aa97c70", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.0859375, 0.59375, 0.830078125, 0.9140625], "student_probs": [0.12208738178014755, 0.24091024696826935, 0.30513450503349304, 0.33186790347099304], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5c6136ddb06c3815b832cfc1e8cd0cfb9f175ed86c4d8e114ec5cd5f457c3492:action", "state_id": "b4df81b937e5c90c3439fa7974a0b3c2a18b197390ec77bb0363b5ef4c28c991", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.09375, 1.033203125, 0.72265625, 0.775390625], "student_probs": [0.11450223624706268, 0.3533812165260315, 0.25904467701911926, 0.27307185530662537], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8639a0d8b5d23b43574ae6d18bbdad3912d2132b7a5c0f2a89795c9e9585c533:action", "state_id": "0ce67266e079ef6bd9348839cdd64e7dc4be4fa80e91439a01a6cfa6acaf08c5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6611328125, -2.73046875, 0.109375, -0.0546875], "student_probs": [0.19527308642864227, 0.024657055735588074, 0.42195844650268555, 0.3581114709377289], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3245b72905e7b613948984f84262349be495b2cd5e371f0448585e49521a0994:action", "state_id": "1a18df772c95466714afacad3af6842f3758c3cbb43f5275df89725a47992151", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.86328125, -0.8154296875, 0.84765625, -1.13671875], "student_probs": [0.04770343750715256, 0.13602721691131592, 0.7176205515861511, 0.0986487865447998], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ba29fdb82881ebf6835fd34f2953e1811278a3b09c65941bb83fa13fd3135e9e:action", "state_id": "fa2c7402f8c6126bcf3a26d251d56e351e2c04b323b50a571666e8a8b6c6b194", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.59765625, 0.203125, 1.755859375, 1.3203125], "student_probs": [0.14454834163188934, 0.09742499142885208, 0.46027177572250366, 0.29775476455688477], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7d94e26204ed4b846c94bae8b07920d603b5cd175f7686e45e39ab7f411b7ec1:action", "state_id": "364bcf639d3e972fea6b891462d2a5698fa0f8edb209c925fbb182bdc90509f0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.212890625, -2.84765625, 0.296875, 0.12109375], "student_probs": [0.2419457584619522, 0.01735616661608219, 0.40281569957733154, 0.337882399559021], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7e5fe7fd01f06c7b52cb81a3c137f64b52cfebb8b8ecd70010eae6f6bcc318ef:action", "state_id": "4d11b834ff6192ad1a023e5f1ec3b18d3f0bbb90d4a85b4317129e1ecb3efa52", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.16015625, 1.31640625, -2.41796875, -2.28515625], "student_probs": [0.010703053325414658, 0.9411396384239197, 0.02248203381896019, 0.02567528933286667], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6b5f4efbaa67a3bb1617597c3f3386ee3d787831ddbd0e6c4cce792037cccc32:action", "state_id": "e0e3760000476cdec6ddc5d76f88efa6493b513b693c6d2ffa4715952fba7d03", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.84375, 0.9375, -2.3828125, -0.7783203125], "student_probs": [0.04848587512969971, 0.7825223803520203, 0.0282815620303154, 0.14071016013622284], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "affe88d42fd993f018a294489afca1d13997575fc51e68cc2d5b96e9c50b946a:action", "state_id": "ace4c1539024dae049230b259bfd331ac163d49ab271f6c29f64287818760478", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.96484375, 1.31640625, -2.44140625, -1.1796875], "student_probs": [0.03287021815776825, 0.8746440410614014, 0.020409582182765007, 0.07207614928483963], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9a86a5508a2cb4a09d1bea33dc0047a5fbfb5d922787e685098172d6b609e22e:action", "state_id": "642e28a7674fb6f1e389646fb75fa99ba8696e1946030e94b966dc9c6ac109cf", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.84375, 0.7265625, -2.51953125, -1.33984375], "student_probs": [0.06159970909357071, 0.8051025867462158, 0.03133939579129219, 0.1019582599401474], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "07411ccf7a3c23d809d22fd23ed3d5769c7f4ff0b7576fe415a8590da64cccc2:action", "state_id": "1e3157c7212420b1f0cc03b657e547086c68c9617e5f89bbc7fe6c6054870dc5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.71875, -3.3515625, -3.421875, -2.71484375], "student_probs": [0.3300279974937439, 0.17527654767036438, 0.16337570548057556, 0.3313196897506714], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8cb5414e7dd4bcb8b1b832321352d77b528efc8d29c62762183b57b9a32a2e60:action", "state_id": "7bf215d6cdfa37bf9342fce3a9e808df95e991113feb491b9261006640e177fe", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.49609375, -0.72705078125, 0.45703125, -0.72705078125], "student_probs": [0.08086632937192917, 0.17448529601097107, 0.5701631307601929, 0.17448529601097107], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d1ff406b042749135f11e146f5a47e4fd9d366c3131c260b7e617110436afc32:action", "state_id": "e6b2fa3d40653cc6e78503e5cd5d40261789cc29d937158af9349e3bd4de3b99", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.291015625, 0.26171875, 1.0703125, 0.818359375], "student_probs": [0.10339301079511642, 0.1796969473361969, 0.4033745527267456, 0.3135354518890381], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "925b6bf19c171e0585f01c142e5d9e1d538959f55e00ea9a5ce4dbb70af29a4a:action", "state_id": "fa6222f4b4eaa7c630533afad6ce33ca3db5bccc333823c368e79e530647e558", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.546875, -3.0, 0.03125, -0.111328125], "student_probs": [0.2265249341726303, 0.01948665827512741, 0.4038243591785431, 0.3501640856266022], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0d9edac9dd32474799b22a6a44520724e985a9f39d175c181171158803cd6e61:action", "state_id": "25442ca1975b55f075032986a8910c573794dfdce7548f001167a23c0c7d006f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.29296875, 1.18359375, -1.2734375, -1.55859375], "student_probs": [0.026174990460276604, 0.8467172980308533, 0.0725543275475502, 0.05455336347222328], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "16b2c0f22694cc10a9359578bccd8359138150c0bbba39e3a3f7246285b57887:action", "state_id": "d72759aea4b433997084c0d8044dc1a1c2974857a9d786a8f83329ad6106406c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.81591796875, 0.55078125, -0.4609375, 0.33984375], "student_probs": [0.10498712956905365, 0.41179966926574707, 0.1497276872396469, 0.33348554372787476], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "93b88e31d911c980818c11fa8c3457a64ab4f3e75c22f9619ba31ab983288f64:action", "state_id": "ad51fc356765ce4194d00b47b2f2fd626c73e51048ac20d354915f09fbab4c54", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.38671875, 0.796875, -0.86767578125, -0.42578125], "student_probs": [0.0705580934882164, 0.6264256834983826, 0.11856713891029358, 0.18444916605949402], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "78797d2d1dd1cf1cc1f86bcfd9a716ff97640e1507d454e58147b77984b40bac:action", "state_id": "f8e1d49f082809b9fb4daafdc9d958e9c22e328575c6e10e0582dedf19011be3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5390625, 1.119140625, -0.853515625, -0.921875], "student_probs": [0.05233084037899971, 0.7467937469482422, 0.1038692370057106, 0.09700605273246765], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1ede4c4cc73ca4bfed198c45e42b79bb4ec77092583ab549537b2fc2e3128fb3:action", "state_id": "04b640a02941caf46e9e39f3e2c819b47ecf4cfee09665a85fc5355c41a6c6c8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6171875, 1.1171875, -0.99609375, -1.244140625], "student_probs": [0.05072735622525215, 0.7812070846557617, 0.09440169483423233, 0.07366384565830231], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "df7f41841f20092fb558908362f76f7222970daaaa0ddec17ea27c1e05c39193:action", "state_id": "5be7530e7980662216c5a42acf6d31758aeb6c569cf3c38a231a272587a864eb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6591796875, 2.20703125, -0.412109375, 0.23046875], "student_probs": [0.04487351328134537, 0.7884418368339539, 0.057450175285339355, 0.10923440754413605], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "01832df12dc9c1aae6446c659583f04052a4e35d2156c2c1978d339d00061dc2:action", "state_id": "0d174bec235db66d742f39d3b6c1a49d7b22379b6a4509fde8c909477ee9490d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.095703125, 1.609375, -0.684326171875, -0.458984375], "student_probs": [0.05166718363761902, 0.7727077603340149, 0.077960304915905, 0.09766483306884766], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c5e6071e54dd1ec6614c90299a0876495e23a30ba16c7bed26fe07a32224ba02:action", "state_id": "b7059a25b02b090667e71b7ea8182ad68a18405c2b5dd097b0811c0d301bcf99", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.95703125, 1.603515625, -1.32421875, -0.900390625], "student_probs": [0.024424782022833824, 0.859323263168335, 0.045989394187927246, 0.07026255130767822], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "43341fe3b7876ab0e46ec02ce7473a7446119b37478e67ed6d81bd85a04e18d3:action", "state_id": "e5872bf43b4dbcd3b7a16b99cef7a0395c91ea1860d29708bf20904445b23f1c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.46484375, 1.44921875, -1.21484375, -0.77734375], "student_probs": [0.04404454305768013, 0.8118081092834473, 0.056554313749074936, 0.08759303390979767], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f0fd3fe8e4853dc608079e02a852f17803e0f4b040b0ed1e989c74435c8df614:action", "state_id": "f8ad67adf5188bc7d13874140d2dd6aa36881b316cae2112389c660e2e122b4c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.59765625, 1.716796875, -1.63671875, -0.880859375], "student_probs": [0.03172900527715683, 0.8727807998657227, 0.03051348775625229, 0.06497666239738464], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ea1e1a4eed2a93a085c959b49ebb95c4ff377a32c16fd236fa45d4222c708cbb:action", "state_id": "a0bc839821c2dc2cb53ffd583d25223dd59e1c63fa1655c53fce6cca6a37ebf2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.88671875, 1.720703125, -1.8828125, -1.48828125], "student_probs": [0.024774378165602684, 0.9134529232978821, 0.0248713418841362, 0.0369013249874115], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bb7f3b3df813a19711478d46fa5d69f6b6b51a15222656497e09d107703a74eb:action", "state_id": "db7ac4cec039c708b75da793b9fd9bbbf62edba57a55ac223a651ef808c19144", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.21875, 1.4296875, -1.921875, -1.703125], "student_probs": [0.023565495386719704, 0.9052590131759644, 0.03171084076166153, 0.03946478292346001], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "65ce116b4bf41c328f796e170219c207bf2c29ce8eddc717c8d6fecb968e3935:action", "state_id": "0967d1737a9d99da06135bb2009882b4135a636866ec78842f5dcfc1f3889690", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.984375, 1.5625, -1.28515625, -1.453125], "student_probs": [0.02536916360259056, 0.8804290890693665, 0.051047325134277344, 0.04315439984202385], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7c81cd56f01a7e8400bb9c080d37d2cd15485a136e63aa2b67dd471c89158f13:action", "state_id": "fcb77056248cc2a326ce4b5c441be624808bb1139404b73fa75c7c3a3384d897", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2734375, 1.6328125, -1.828125, -1.22265625], "student_probs": [0.018137942999601364, 0.9016766548156738, 0.02831292897462845, 0.05187242105603218], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2548e1613a9ebb2fb2fee0756375a6ba242b437773157f7e0bbd2eb180bcb0f1:action", "state_id": "a54a6d1b76cdfc5b5c7367c58d5fe7aeb54c4c60e67c2b823ae14f76d7e56956", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8828125, 1.49609375, -1.828125, -0.94140625], "student_probs": [0.02944774739444256, 0.8639575839042664, 0.031103018671274185, 0.0754917711019516], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5437b1b0870538a9cf95c64c9a64bf42bd668f8bd732926f689db1d462979d6e:action", "state_id": "c475eccfaf205861c0fabfa03299af8b9f0ffbf8438168cb6a1054abf80b4549", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.77734375, 0.71875, -2.2109375, -1.150390625], "student_probs": [0.06387705355882645, 0.7751479744911194, 0.04140354320406914, 0.11957135051488876], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b5cce310617a0de7a8484739703fa9804b8c18123d92da8d44d59657c140eb24:action", "state_id": "5435eed455c4ec1b3761cb95508871ce3d4613b51a92790231ab3ecb99ce5dde", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.5703125, -2.95703125, -3.1328125, -2.30859375], "student_probs": [0.28183093667030334, 0.19144272804260254, 0.16058242321014404, 0.3661437928676605], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9474af04af8d37cd0d30b719415542ec5191208fd8b44cf35f0f865e52aeb803:action", "state_id": "d749a251ec6c403b8ea149d8d038f87bca94fed123a00377e22d3f43c695a8be", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.44140625, 0.8046875, -2.76953125, -2.73828125], "student_probs": [0.013367187231779099, 0.9334587454795837, 0.02617168240249157, 0.02700246125459671], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0e174a08374908d94469978bf974e9fecc17ce511b73f2dbf570e9bc35c2f65:action", "state_id": "143b53acb0d832874a992418f1e4f2dae847095837195e76a31205b7ffafcde8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.484375, 0.5546875, -2.30078125, -1.5390625], "student_probs": [0.03896995261311531, 0.8139129281044006, 0.04682347550988197, 0.10029374808073044], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "809d7bc7c89856ed1088d5e000d22b62f0e7e439f889d874a9868c251a2639fa:action", "state_id": "ae1eb35b7782ea2c3dfe18610512bb943740272f645ca933244881a5507fa4e8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7734375, 1.330078125, -1.3125, -0.4716796875], "student_probs": [0.0350416861474514, 0.7805930972099304, 0.05556068569421768, 0.128804549574852], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1431413b8ed5640f2867146f808b88260ee05fa6409bdf86941f425455bd0899:action", "state_id": "5e627c2c8766d52cb1a4a6cf61e2ed7f92d5c8334ec9ab857ba922c23eaa379f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.00390625, 1.51171875, -1.53515625, -0.8076171875], "student_probs": [0.025289081037044525, 0.8506473898887634, 0.04041183367371559, 0.08365170657634735], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "783ba4d30b6f86a58d827849fbbe271e205edabba15bfbe09b82d284b35e318d:action", "state_id": "392ada898f316a4b07c46608d751cff098172e88fbe200af6ca5f474a16fd2e5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.75390625, 1.638671875, -1.27734375, -0.8828125], "student_probs": [0.028783118352293968, 0.8560828566551208, 0.04635603725910187, 0.06877792626619339], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "40feab8875b9c462422e8ad694fa55bd40295aac09d7a40a57c1aeac87df1f5d:action", "state_id": "f5ebc252ee1342ca29faf526d61ecd353bb1d7e90576c5cbdd891c2083dfa1af", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.91015625, 1.75, -2.06640625, -1.130859375], "student_probs": [0.023308556526899338, 0.9059433937072754, 0.01993686705827713, 0.05081123486161232], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "471a06c38e47d066410189554e13c373559aa25ba862553f77597049694bec81:action", "state_id": "c25a28d93c75beb7df48199816db39849f04cf58b6c45341cfb601d578013b6f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.52734375, 1.453125, -2.37890625, -1.6875], "student_probs": [0.01723598502576351, 0.9228512644767761, 0.019994091242551804, 0.039918627589941025], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "530c8112446ba895f6c141778aef45e51e503f14eb5216521768c04af9ce4e55:action", "state_id": "961bf7e69d8671f6d8646a3888f5870802183fb0e74ae369a1b0697f46d09f0a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0859375, 1.84765625, -2.109375, -1.6171875], "student_probs": [0.018293213099241257, 0.9346048831939697, 0.01786945015192032, 0.02923247031867504], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c6dc12f6200f3130f00c76183bd1fa0b9ba7ad9621aecf1474100952a5e40934:action", "state_id": "ab5681dcd687fb494afbb71de597b4719530df78c6494776156a66753ec8084c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.09765625, 1.62890625, -1.87109375, -1.79296875], "student_probs": [0.022150076925754547, 0.9200274348258972, 0.027782421559095383, 0.030039958655834198], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "192eb57228262d90d002a04bf377f549a0deec604ed678a14b989b0bbc991bd8:action", "state_id": "d530c0ac32370773c2d3b582abc889fff553879bbbb22ce257ae4d85c32a260d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.56640625, 1.6953125, -1.1640625, -1.1484375], "student_probs": [0.03321315720677376, 0.8666757345199585, 0.04966447502374649, 0.05044657737016678], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0ee5a8735d7b80b1e77347230b39b48874e8b63cf4a0f871db550fa91f2ad1c:action", "state_id": "28792e9e87b08d269f0218f03d52ee64cb06b1d747c7bebadbbc8afec65c1f78", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8515625, 1.77734375, -1.71484375, -1.01953125], "student_probs": [0.023743903264403343, 0.8944705724716187, 0.027222517877817154, 0.05456305667757988], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "793600ecd101de2d724e908d2fb61fcbbce18d6cdae058455c86fd23ce2fc704:action", "state_id": "17a6c53784a6a36ee672188eb015e3ad11dd1105d6f537648e95c0505ab7902b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.71875, 1.291015625, -1.33984375, -0.74755859375], "student_probs": [0.03939424082636833, 0.7990193963050842, 0.05754261463880539, 0.10404369235038757], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1c06be305dce14ed1bc2c087187370013b52efbd457e1907674fd090dffe82dc:action", "state_id": "5534daa5ccdf419652fa6726887e25d47a29d24ace77c71e26bb0e18eafd093b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.765625, 0.80078125, -2.328125, -1.60546875], "student_probs": [0.06344199180603027, 0.8259483575820923, 0.03614816069602966, 0.07446150481700897], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e385d53eccc1284966590c7b60434166e4801e014c745afa2d0c4a1deb5598b0:action", "state_id": "dd2f1a957cc1198b34c5de24b558f823bf300645049b852e5a715d9227756c82", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.19921875, 0.8359375, -2.53515625, -1.875], "student_probs": [0.04183777794241905, 0.8704026341438293, 0.029900111258029938, 0.05785954371094704], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "46161a2de00747a5b658fc5b9721d4fe0ffbb4ca4fc376798276d9b67884d6a7:action", "state_id": "359589358d4ef4345beb31bab9f14c667048ed93e2e758682c00a854b0ec2b8e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.84375, 0.73046875, -2.46875, -1.34765625], "student_probs": [0.061354897916316986, 0.805041491985321, 0.03284091502428055, 0.1007627546787262], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7f72a4e109362ddddedfd340a7bc5d51606c4b2b57220169602235cb25b3b515:action", "state_id": "d1514c1225684d9752be495dfe89118a4a1a2cd0379c567e52b2e3604f3afbea", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.98828125, 0.63671875, -2.4921875, -1.51953125], "student_probs": [0.05880022794008255, 0.8117120862007141, 0.03552510216832161, 0.09396249800920486], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c097dc7ee59df5058c44538506594d107060bf884d4a627bcc7871301c28ade2:action", "state_id": "442aea0cb2ddb61b712bd607275b6e1508d4de7bdf4830ce4f3a78ba45bccc36", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.201171875, 0.54296875, -2.0, -0.884765625], "student_probs": [0.1170545369386673, 0.6696670055389404, 0.05265766382217407, 0.1606207937002182], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f8069bde2cb98d3667b4955431ba1293894b1c543b2a3dfa07e259a773b8fd36:action", "state_id": "b5d657aa444b28bbae1745f571d98c9ab03485db7a3f2815df506170682e3c2f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.64453125, -2.9140625, -2.0625, -1.24609375], "student_probs": [0.2916437089443207, 0.08194117248058319, 0.1920132040977478, 0.4344019889831543], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0981651b69c41346b083e037f20f62aa01d43a42cc3ef933a58c311c763cc103:action", "state_id": "9772e6aef5886637c6ed5e86dab7e6b3673f9fe953ae406716cdef13b676ddc7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.05078125, -0.544921875, 0.974609375, 0.2890625], "student_probs": [0.17232443392276764, 0.1051342710852623, 0.48047229647636414, 0.24206897616386414], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0186dd09d24e53ba5e431d4d15a0bf1625b1e743b97e64c8893d2d36701ea94d:action", "state_id": "59c838bdbc8ff367eebc71d4033de403163a54e8d69b84f39d9c38e0bdd75c95", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0703125, 0.203125, -2.390625, -1.4453125], "student_probs": [0.07514898478984833, 0.729901909828186, 0.054552312940359116, 0.14039678871631622], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2aa727f20c6a1beb817369869744ebf9abc4dd92fe2ea9650b1d727999807033:action", "state_id": "1bd3e7cf19565e3ca9a39d336ee9c31008703a6f73fd1d3aeb57945f10740035", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8515625, 0.1796875, -2.55859375, -1.19921875], "student_probs": [0.09060637652873993, 0.6907476782798767, 0.044678542762994766, 0.17396746575832367], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "146696b6f611b79238cb1e6b9e59bc7c69a2292c409250d104f29f02ac947ae9:action", "state_id": "0aa50088a4f92bf09c48c5aede7bdf21363d8da1f22ff83612338f4268072969", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.19140625, -2.80859375, -3.109375, -2.03515625], "student_probs": [0.3217599093914032, 0.1735764592885971, 0.1284881830215454, 0.3761754631996155], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a7ab091890f867616b9a2c0caca17289c4d6ac7887220810c5558f6967b41efb:action", "state_id": "6f103aeb14ae81cfb0868b3c3a3dc7e37217e99681d701ab426338d773bffe48", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.06640625, 0.18359375, -2.94921875, -2.83984375], "student_probs": [0.034283023327589035, 0.8841708302497864, 0.03854544088244438, 0.04300054535269737], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fc8bdaa956f922540b64b0a6b30367d7b2526c93416b28adcde939a0b43cfe89:action", "state_id": "ea90d22481afa3ba814db07fd3f8b031a9dc587af72c6a3b5138da51e9b529fd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.23046875, 0.33203125, -2.36328125, -1.71875], "student_probs": [0.06056207790970802, 0.7853809595108032, 0.05302992835640907, 0.1010269820690155], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4681d18f71968c6cd53331c744312c7fcfa66619e6a30b6225e8b97050220911:action", "state_id": "29ff38e2449323e81d0d45c5ee6678336631e7853160583a38b1c4c2820690b8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1171875, 0.69140625, -2.30078125, -1.39453125], "student_probs": [0.04883110523223877, 0.8099408745765686, 0.040640849620103836, 0.10058706998825073], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0fc1a5a77ec8da984b4c79dc88639959381dd64cb8bc37ec9c1eba478027ef55:action", "state_id": "98aaceb8765e298fb7950e7d7d5be8fb1ffca70f1431aa90186d6d930e335d00", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0390625, 1.25390625, -2.25390625, -1.33203125], "student_probs": [0.0325126051902771, 0.8753262162208557, 0.026226861402392387, 0.06593432277441025], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c7ffcd3fe58844e876473f87055f5cd796a2424741c8f830e6c78fad9aec75b9:action", "state_id": "4173d4b17a3b0203317cf9ee3a190319514bf7923afe2eb1c7b94821216f0dd2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.08984375, 1.66015625, -2.328125, -1.3359375], "student_probs": [0.021535782143473625, 0.9157246947288513, 0.01696978695690632, 0.04576968774199486], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cd6d2f2dd8361fdb4941b5007411250f0f2f46c16a9fa2efed8dcbf0fe3f40c7:action", "state_id": "47ea7c257e8efa865128e75b586d65aaf62fb14c329f94016e5b557dc2782749", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78125, 2.11328125, -2.203125, -1.33203125], "student_probs": [0.019100036472082138, 0.9384424090385437, 0.012526108883321285, 0.02993142604827881], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aefb604b572efe27d8cd47f2533a969a8d7ac60f625808e2eee45ce9bf45a6b6:action", "state_id": "cc80d9846ce568721c9e81bb023ce83a1cd15548cf3bd5a55a9550db883aafb2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.61328125, 2.07421875, -1.3125, -0.7978515625], "student_probs": [0.022443706169724464, 0.8965107202529907, 0.030319513753056526, 0.050726067274808884], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fb699c4ed1129bdc80bab15ac9aba9797f2b2a770833df3d22b8660e147c53fb:action", "state_id": "672442d76b809d4b92d82d6cf03c2283fa73a1b8720d8541c6abfc0eb77dbae2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.80078125, 1.830078125, -1.87890625, -1.18359375], "student_probs": [0.024082576856017113, 0.9090026021003723, 0.022272741422057152, 0.044642042368650436], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ec8e937f69d7f3f0572dd5893c19e53456cd5768eee926ada48411070bf0a76a:action", "state_id": "2101e9aedd6b719f8ef4508503251c48bafd5e510cac7480b9ba6f86feab0c65", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5390625, 1.740234375, -0.9921875, -0.77197265625], "student_probs": [0.03180820122361183, 0.8447333574295044, 0.05495963990688324, 0.0684986487030983], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2fde6d0478b98fbb5ea50682663ec614bce1ab69bbfe55670029b9137102b956:action", "state_id": "fb3353477f1bc364cc4c3d81683deda52bdf00cdb8b6a91799bcaba256343c05", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.072265625, 1.462890625, -0.7452392578125, -0.310546875], "student_probs": [0.05831858143210411, 0.7358872890472412, 0.08087842166423798, 0.12491574138402939], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "80a7af8210d4e0dc4b5f4ef4c9a594190bf053c1871c83182419cf0ffee62ff2:action", "state_id": "1eeac0c669e8241f8ac772b984b45ad1223bd50a648a1796f449db3f5afc9454", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5859375, 1.841796875, -1.7265625, -0.6700439453125], "student_probs": [0.028429610654711723, 0.8758244514465332, 0.024700075387954712, 0.07104580849409103], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "94ebbb3a81f610a7b6e359c3f34b8482b6de197db5469aa3174f6663f087c051:action", "state_id": "fc0c0209fee2fd9d1aa6b28978ac2e635fb135bad185956dffae67261ffad82b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.07421875, 1.12109375, -1.82421875, -1.65625], "student_probs": [0.03543497994542122, 0.8652443289756775, 0.04549941420555115, 0.05382123962044716], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d6cc8f44e05d36f403e9f9af4b7f6f73f2a0a5019328a21e6d1ce6fa01d1fe69:action", "state_id": "8027aff256c761bfdc854eb6af514e14424268f5a108c8737957cdde57b72df0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.23046875, 1.3359375, -2.20703125, -1.69140625], "student_probs": [0.02555760368704796, 0.9044627547264099, 0.02616368606686592, 0.04381592944264412], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "781ae6419bd1c6c29dcdac71d78284d77ae7728dc58ea1bbba0ed4d4b5e2a512:action", "state_id": "c31693296fc97c54a4704de3b477388f30663b749c3c142e0e688fe1ecac407b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.31640625, 1.625, -2.03125, -1.6171875], "student_probs": [0.017910519614815712, 0.9222298264503479, 0.02382045052945614, 0.036039188504219055], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f7ac509255e5befb1aa6482e2e2b93335977b1ec333fc601fe460952b73dacd0:action", "state_id": "fe59391841fb5fdc9cb553315863fe2b1477a1d35ad9912618178851544d137f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1328125, 0.671875, -2.28125, -1.7578125], "student_probs": [0.05040587857365608, 0.8328015208244324, 0.043452586978673935, 0.07334012538194656], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "35e4a1de3956610da3e9a0daf8058007a2ada91f2f04fd5996f08e21c8adc863:action", "state_id": "75abd0cdc7f6edd7ab0638738e22476415331e82e4d3b1577295e73c2e4505e6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.125, 0.78125, -2.25390625, -1.65234375], "student_probs": [0.04593183100223541, 0.8400054574012756, 0.04037667065858841, 0.0736861377954483], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e782caddf4c353bd86059a9f184312473926bf2ab63379a046fb5faf51c381c4:action", "state_id": "981dd08a87a343c2e58045e654180f2d710af8032cd5af116eb5c2812011450e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.4921875, -3.4296875, -3.3359375, -2.91015625], "student_probs": [0.19907300174236298, 0.2119120955467224, 0.2327399104833603, 0.35627496242523193], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1c85c6c8e52b3454397890d5e9d4ab43a487dd5502a10be2b76577ad66cb3ac0:action", "state_id": "188680e5cb89bc8ec65e43877aa8772da0e52bf55a7dcc6db7992acdaac6bae5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.91796875, 1.158203125, -0.189453125, -0.802734375], "student_probs": [0.03188995271921158, 0.6912232637405396, 0.1796133667230606, 0.09727338701486588], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c3d5a43279ae36e3081b0608216e0ed2b0a9418206d38bee4a600bf48136869d:action", "state_id": "f48e7e12cec8109e73539ce8793e6080405291f5d1274c17755a46ab805e3127", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.06640625, 0.6328125, 0.3515625, 0.39453125], "student_probs": [0.06707629561424255, 0.36688536405563354, 0.27693960070610046, 0.28909870982170105], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "59b5694c13115dcae7cff6ae64a4b4b916c482dc8ed74a32654ff693d1405040:action", "state_id": "d2288bd1752ede80a9284a42dc9a54d673b0abed1f5faad1034bb68db8bace48", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.587890625, 0.705078125, 0.611328125, 0.669921875], "student_probs": [0.08711689710617065, 0.31741803884506226, 0.2890124022960663, 0.3064526915550232], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "02fc75f241224ddb871bdddb7038423663f81d0fe30c6b7e927b637161ef55dc:action", "state_id": "7fa0ac70ac40260d420c768832edd4b5061d64b8e3754a232beef7fa231857ce", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4208984375, -2.6796875, 0.00390625, -0.04296875], "student_probs": [0.24431784451007843, 0.025525575503706932, 0.3736332356929779, 0.3565233051776886], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "248923e9d965121fa8c35c46fb067de86b84aed83198b00b8956dd991f37e3b0:action", "state_id": "8fcd6c87cad8078863b0da93fc1b689e2959c595fbe057a24dd2c09f418bc7f1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.5859375, 1.046875, -1.78515625, -1.92578125], "student_probs": [0.02326587215065956, 0.8798928260803223, 0.05181962251663208, 0.0450216680765152], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4d03e5b19e4fa9448d643a03b356961200675cea1eb1b985bcd92508b97e03b9:action", "state_id": "fe97888b22e3c37b839177e581032bd1180a5bfdd1a5d0c93bb58d319cfe9643", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2265625, 0.42578125, -1.13671875, 0.17578125], "student_probs": [0.08788955211639404, 0.45871296525001526, 0.09615146368741989, 0.3572460114955902], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0a35bbffa4ed228a6a3a569d93b0e9cfee71d7f96dd12ad8886a0ea433040b30:action", "state_id": "82445788b800b85d06810020ce98fc9595708083dc59e164394d654f9c6a600d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7890625, 1.26953125, -1.4296875, -0.765625], "student_probs": [0.03771768510341644, 0.8032956123352051, 0.05402808636426926, 0.1049586609005928], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "03a5ed4daba9ca7dfb6626d3baaee20f13d1e8e800e47c78c02e39babb2157cc:action", "state_id": "16e3a377fd8b6c149d7ab4c273bea6cca880a7e82178fafe59b78b98e9ddb004", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.85546875, 1.37109375, -1.85546875, -1.078125], "student_probs": [0.03404998034238815, 0.8578179478645325, 0.03404998034238815, 0.07408203929662704], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8f408c6e636c716d6f3c4f3122c1dda0e847f37de3cac06c3770094a096a9827:action", "state_id": "20ec15ff0c02e83687b17c956602cb98bab02b0d7682baa4724e9f7daa705130", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.921875, 0.796875, -2.421875, -1.58203125], "student_probs": [0.05502784624695778, 0.8342968821525574, 0.033376071602106094, 0.07729915529489517], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "55038a7933bd94dfb65eaaa90ac9d0e8679706979e41730b59ac5e9dfaf3c237:action", "state_id": "e10716777743533bc62f91462cd816504580a118a9ef0bd3476f29818b31704b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.6953125, -3.1328125, -3.2890625, -2.56640625], "student_probs": [0.29980653524398804, 0.19356964528560638, 0.16556888818740845, 0.3410549461841583], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3c9f311bd3f42e77aaa7dc0ea3316dd86c57339b41473e807823c9f167c06478:action", "state_id": "2a6bc90839f86d8619ef7576a5a873b2851f25e7d55975790d5ae21f8832ab67", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.05078125, 1.0, -0.03515625, -1.1171875], "student_probs": [0.03107433393597603, 0.6566581130027771, 0.23322583734989166, 0.0790417343378067], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e4dae3d04fde4a0f971538570b5329e544a7897a770282bd2d63fa96cd063294:action", "state_id": "2d857184ec757911a5c176d4a11e4622bfadbb8e706bb76284cfb66049e78571", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.162109375, 0.51953125, 0.00390625, 0.2890625], "student_probs": [0.07219351083040237, 0.38799434900283813, 0.23168200254440308, 0.30813008546829224], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0c291ee9c69ffcf2b7e8682386a4c92cea382e9e01392c0fa1280d10bd8c9d78:action", "state_id": "393f08c82c477f546d1dca43969d6de92411bee989d81f22b1d95e30cc1a43ba", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5859375, 0.890625, 0.73046875, 0.927734375], "student_probs": [0.07325338572263718, 0.32069385051727295, 0.27323460578918457, 0.3328181803226471], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e0bf2c19f50e7b67b9b94a6b3648cbe930905adfc34de279fff197eaf3e950ea:action", "state_id": "409ea34f114ea5126be568110ef26187a9ae55d5a805406452a6f635e7b50d4e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.328125, -2.48828125, 0.13671875, 0.08203125], "student_probs": [0.2372971475124359, 0.027362046763300896, 0.3777213990688324, 0.35761943459510803], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ad38cc944a145328874aa21198ec2378bdf2e98c73e6fae5b85b990e05446ab5:action", "state_id": "c1cc8851fabba266c826b0e770486dc00d9cccf1b22ad2840e9e32e80de69e9f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.34765625, 1.0546875, -2.6171875, -2.70703125], "student_probs": [0.01154524739831686, 0.942577064037323, 0.02396855317056179, 0.02190903015434742], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "59f94816927886ad24eb7b25c58b88ef545684bea87dd59af9ad03f053c79e61:action", "state_id": "a7658ebd46cd5d5cad7882e33af3e659cf11b00deb30002b46b7e3a6ad907670", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.109375, 0.6171875, -2.4375, -1.26953125], "student_probs": [0.05176907777786255, 0.7910455465316772, 0.037287868559360504, 0.11989744752645493], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "145b4a048ef339e0e0b4d822dfcc8e5de471d70a25d789453ec77e1af93182ca:action", "state_id": "d4d0afd647485dece785f869f4f3a07842b5b95e4b2db6b28f660d4583fd812f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.87890625, 0.19921875, -2.5703125, -1.609375], "student_probs": [0.09259520471096039, 0.7397870421409607, 0.04637827351689339, 0.12123958021402359], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "df0d31d915b00b3ad9ba86b22c7b8e99cf81cd553ad1fc2b8b60affc93e40a35:action", "state_id": "bd10b83198f05ddec8f9fb374846434b27c4abe777bc67cc31e29ebec211df10", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8203125, 0.37890625, -2.6328125, -1.4453125], "student_probs": [0.08391580730676651, 0.7567498683929443, 0.037237413227558136, 0.12209678441286087], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d3f1c3806819a7ca909ef732769861f092653f6c9d555876b14a88b2f80e7491:action", "state_id": "9a206043216dd99627aaba127088be2886804c342d25cf7989e0dfd45f13d939", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3515625, 0.671875, -2.53515625, -1.23828125], "student_probs": [0.10009603202342987, 0.7571547031402588, 0.030647048726677895, 0.11210224777460098], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "abfb9203d9586933c1286ba1d7928ce9bfcf087be6ec94d9017e6ee871ae0804:action", "state_id": "0fac01ba7071abf1f6d0c7de12aabed96ec476dc58b0e6c562d83d2051cbfe79", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.80078125, -2.25, -2.1640625, -1.62890625], "student_probs": [0.2840064764022827, 0.1812320500612259, 0.1974954903125763, 0.3372660279273987], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "242d9210f7042c800ccc9c74b36997ec7fdf2c14669ed2af1f64ca49fa2d4c4b:action", "state_id": "fd1398b71e49671ca2e2a536813e8875b8e76ad94d1f54690962a0cdec1f5602", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.34765625, 1.18359375, 0.28125, -0.52197265625], "student_probs": [0.04773053899407387, 0.5999351143836975, 0.24334441125392914, 0.1089898869395256], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "15447bfc7f75294243d11759b6115f62eb70c8ab78bf8a5e29c4b3c69a3258f0:action", "state_id": "ecc435829864efc7a561b3ad2cb1e3e0fe93ea113177890ef6d7c9baff9a1c72", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.63623046875, 0.580078125, 0.43359375, 0.2734375], "student_probs": [0.10232196003198624, 0.3453066647052765, 0.29825490713119507, 0.25411638617515564], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7567fbb63cc2114d63a63fed8ab57066c6a5bc5e378f9d70f1c3dcfef96f5903:action", "state_id": "e1ba9550adb2e9a6566fd3ca864517d4610da42c345b4183c793d8e0c6918f6e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.16015625, 1.34765625, 0.10546875, 0.33203125], "student_probs": [0.04701421782374382, 0.5772425532341003, 0.16667987406253815, 0.20906339585781097], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "610bfbf2b75abc2b3a90b76d639278bfe25c2c9052b7b2b2c9e0df28f3796169:action", "state_id": "9fdec8cfb9bef77bed4de05c973323079998320c0732a796415ceea599969826", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3203125, 1.701171875, -0.296875, -0.123046875], "student_probs": [0.036211512982845306, 0.743122935295105, 0.10076737403869629, 0.11989816278219223], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d15a8af6d754f2e86508ef7c702b3396bd67af57971dec463864a3abbf6473b1:action", "state_id": "23e594442bfdbd4c3fcf4ef04fbaed2b5e8a20d3c59b3e17080c09e06193a3f4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5703125, 1.849609375, -0.6934814453125, -0.4794921875], "student_probs": [0.027065787464380264, 0.8273206949234009, 0.06504644453525543, 0.0805671364068985], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "622f0408537c8565e3519d3c22a2103e1174d3baa8cf22d9b6ca201d81319bc4:action", "state_id": "b7fc63e116e4f048d75f725668b5a3a797e6e4d153d03232aea57d746985361b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.48828125, 2.181640625, -0.4599609375, -0.14453125], "student_probs": [0.021331695839762688, 0.8372443318367004, 0.0596512071788311, 0.08177275210618973], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f76d521927f0b2afa0ee4b1cf4bea7577740233192122db1d66c4c505fae4730:action", "state_id": "d0044495b315e828efe7e95267c82256a9c2d5dbf5fb21ed7f2760e39e7f1212", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.54296875, 1.900390625, -0.6322021484375, -0.2578125], "student_probs": [0.02604616805911064, 0.8150341510772705, 0.06475670635700226, 0.0941629558801651], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "064a0996d29c90a9b68d94512e5cf2ba93c71a8f24119c4aeb276e2cd5ec0c8a:action", "state_id": "910190fc0832f6c88f2d28cf8dbdb728fc928c656d7acbbd4de8552e3c757129", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.43359375, 1.9140625, -0.70703125, -0.216796875], "student_probs": [0.028669459745287895, 0.8152449727058411, 0.059287287294864655, 0.09679828584194183], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f72d0b117a11d63e16286e7dce4b2cee6185fce5974a0a15ca1b5c71ba13daf4:action", "state_id": "5c6c99b8193a36b04c566bf222d8a7335e02689d6d68fd9ce4948abdda20a705", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.208984375, 1.626953125, -0.81103515625, -0.4873046875], "student_probs": [0.04631124064326286, 0.7894387245178223, 0.0689467191696167, 0.09530329704284668], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d266f93c94e4531987405a09b1f0b3aacda893584aca3eabccb9b5a439c8bcf1:action", "state_id": "345055961a574800c5fc32ca07df527019a2889c4cf6336178a4f1b557b35ee5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.244140625, 1.7734375, -0.73260498046875, -0.4267578125], "student_probs": [0.03941019997000694, 0.8056124448776245, 0.06573031842708588, 0.0892469733953476], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "89cc16f51a98c440780808215f9f5933b5252991430134fe8e6bdc002674eef2:action", "state_id": "7895009365808993a5e848a4521b36f9a9281cc645a7b6b35aa94e7a713d9b3b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.146484375, 1.908203125, -0.5224609375, -0.140625], "student_probs": [0.0372922345995903, 0.7911381721496582, 0.06960306316614151, 0.10196651518344879], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3e4847fdfc57a3c25799f589f0a073e9ae272eb3e046af8d6da812ea12dd44ae:action", "state_id": "f90eb2b69bb0490b1873eba85770eb60b1b43dc0571e7fddb42231ca4874b892", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.228515625, 1.6640625, -0.39453125, -0.197265625], "student_probs": [0.04141335189342499, 0.7470868825912476, 0.09535318613052368, 0.11614662408828735], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0437aa6bc73e4e9273c13240c63310414ed3323a343fa46e91ac73b81c14ea4a:action", "state_id": "966428f93e8a27507a4e2f35280d1ed163e552eb5cd141ccd23ac15165915133", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, 1.91796875, -0.00390625, 0.109375], "student_probs": [0.03801090270280838, 0.734221339225769, 0.10744032263755798, 0.12032744288444519], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e00961a0322592743c6d0c480c7e765a596fa9fc6c400da12bb4d95c5ecbb4fd:action", "state_id": "b7e75f50b79dcb1e849f389f8bb07a84c7d26c89248e6a36569b0ba3e2dd517a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.046875, 1.9375, 0.37109375, 0.421875], "student_probs": [0.034191928803920746, 0.6761159896850586, 0.14116908609867096, 0.1485229730606079], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3853d4ca05940078a9aa0c01d01b93f8dac0279df233ef33ce190173acfff40f:action", "state_id": "9b8c467e8d513e89506e9df88aa0733b6df1f5f32970509aed1d291b089b3afb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.052734375, 1.8203125, 0.3125, 0.328125], "student_probs": [0.03761406987905502, 0.6654244065284729, 0.14732080698013306, 0.1496407687664032], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7751839a25e3023cead334f376b1de86d06445d3340c23942b21ea277b4848a1:action", "state_id": "36fad5e77ab6af627241dc2194ccca626ba0f7aeea673f2e7bf15cec85cdbb7f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.208984375, 2.029296875, 0.76953125, 0.935546875], "student_probs": [0.06180976331233978, 0.5796025395393372, 0.16444513201713562, 0.19414253532886505], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "01d1e4a6f331046910156649f2141da0980b851940dfa2c6550b19cd5cce83e1:action", "state_id": "90865a768dfdb5d8ff76918e9087a2812edc6886c587e0d2d6a91f6379a83318", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.13671875, 2.0400390625, 0.716796875, 1.01171875], "student_probs": [0.06527917832136154, 0.575610339641571, 0.15326811373233795, 0.20584236085414886], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dfe8fa841c92d12830c5a6efd7a7ff770bb9977cbc5fb7acebd9913171844c48:action", "state_id": "d3087bec55186897f5ca2b2cc904546b5d0b389eac5249e6daecff8f50b9bf67", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.19140625, 1.1953125, 1.53515625, 1.52734375], "student_probs": [0.08798269182443619, 0.240097776055336, 0.33727210760116577, 0.33464744687080383], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "eb0739d4bbddd0292bd744496f7fa31102387cabf0268058e4faeadeaa5574bf:action", "state_id": "3d928f52466fc1ebf212b063ddcfe75177d8b81e635ef68be048d23e9b3043db", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3671875, 0.86328125, 1.767578125, 1.837890625], "student_probs": [0.09048650413751602, 0.14860539138317108, 0.36708423495292664, 0.39382389187812805], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f755ea05834da2f30a73961a61f9331f24f3ee59c564bc28692e7a7d43179ae2:action", "state_id": "20c97629947b4aabc9a622cdecfd9576401d99f29ea027f7512c6b285316d9a8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.47802734375, -3.28125, 0.02734375, -0.04296875], "student_probs": [0.23456183075904846, 0.014217826537787914, 0.38880980014801025, 0.3624105751514435], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2140e3a9f82ce3a55e33888711bffb11fdd2ea45a7a59f1510174d6cfa969eb3:action", "state_id": "67bb63e715c1341738978fa77cda81458d0f60218f685d836a377e4090a0560c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.32421875, 0.6796875, -2.54296875, -2.75], "student_probs": [0.016730301082134247, 0.9170186519622803, 0.036542341113090515, 0.029708709567785263], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a2bb7e0baea20b9773978b2dabed2b96ed7f7566f3258705d69cd8df8b38eeb5:action", "state_id": "1f5e7d9c16e47745c1a802f4be4ff2a18d2278e32690bbbb0e63ab5f21ddab6c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.49609375, 0.3984375, -2.01171875, -1.77734375], "student_probs": [0.04395593702793121, 0.7945047616958618, 0.07134753465652466, 0.09019172936677933], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c35a7ad1042d79d0dd5893ae843b8ceab0e93a971901543844015f25cb551fcb:action", "state_id": "9cea5983b2990566fb4039957c074a9b113cb991c01b3cf4e986f71f03b56dbf", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9921875, 1.29296875, -1.5703125, -0.8916015625], "student_probs": [0.031013617292046547, 0.8284716606140137, 0.047290120273828506, 0.0932246670126915], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a7501de18f8bfb07955f36daa5f7fe1d23216f6e956d94f3729126df2db7cfc5:action", "state_id": "6c6b6357f029cc6a114c85b6b182f4677e600cb51a927551ef0f36f83d465388", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.05078125, 0.91015625, -2.21484375, -1.625], "student_probs": [0.04406150430440903, 0.8510951995849609, 0.03739451244473457, 0.06744872778654099], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6b4028fbd54f3ea9e4d9185751f311f75a2b2e5b014e06a38e14c083a5581f19:action", "state_id": "6c894cbcced8908b39ba682fbba4a4dcf089663256b52a80ee15b13eae1b35e5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.09765625, 1.578125, -2.69921875, -1.62890625], "student_probs": [0.02346017025411129, 0.926195502281189, 0.012855112552642822, 0.03748924657702446], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d3ffde45bcd403a056d17a6807f9b591db3f2f050c4d8d61819edeff16cff215:action", "state_id": "fab02cce38c66a4c98382527f8ff0cee3d2d3cc8fa3b0b71473f1ef2a841af2a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.046875, 0.8671875, -3.07421875, -1.73046875], "student_probs": [0.0472552515566349, 0.8709863424301147, 0.016915325075387955, 0.0648430734872818], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8d433a4bc50f623a942fe9cf3f71f44da7d311f7d75a4da7020f61634c02a829:action", "state_id": "2fc0163b4b17c33e6d0e6208ee7a4613e9d802d22aabe4a712e18531c66f18ae", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.51953125, 0.375, -2.96484375, -2.140625], "student_probs": [0.047222524881362915, 0.8535484075546265, 0.03025188483297825, 0.06897728145122528], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "163a1508aab11001d80433898320af1bd95a28a9c3fadb54bc76caa846b11e30:action", "state_id": "71f2a5c50064fbd5e445779cc5c63141bdc1a723b1a845df58e5fe745e041b4a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.69140625, 0.2109375, -2.71484375, -2.13671875], "student_probs": [0.04558919370174408, 0.8304888010025024, 0.04453311860561371, 0.07938887178897858], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b3f51a73944d83b531f6f89d5ad754a2d2019727aaf297b04c08b4794437c2ba:action", "state_id": "878f02ecd4b3f3df199ea4a2c4ec2f525c89943e70c8a510fbef49a2fde62db6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.89453125, 0.84765625, -2.1953125, -1.58984375], "student_probs": [0.0537133626639843, 0.833679735660553, 0.039760760962963104, 0.07284612208604813], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "77e0dedf6fc106cb74bec9783e5f56f218f672a5fb403418d4ff25a017d2b111:action", "state_id": "b7949a0dabefbe732d6e3fb1d6607a15ce4d2f19344c4e1d07cefc577cce7ae2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.01171875, 1.6640625, -1.9140625, -1.6640625], "student_probs": [0.023256994783878326, 0.9181742668151855, 0.02564278244972229, 0.03292598947882652], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e2e134c3cabcc69b0094b51ce52ff48526de89b3e2fc441c6501ae88649b8918:action", "state_id": "d86c7260fc64dc2a30c1377d2c5f55a4eab78b30ce7f747a549cdca01e3677cf", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.078125, 1.796875, -1.55859375, -1.57421875], "student_probs": [0.019040685147047043, 0.917431652545929, 0.03201195225119591, 0.03151565417647362], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fde7c9936e0945c85c85ea0b1e33354f37bae83e90ff874aa496fafa9a267ca5:action", "state_id": "00f8fdf8e624a545e5eb62d9abbfaf4d31a491e0ccdd293eb6eba23ea7029956", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.17578125, 1.29296875, -1.78515625, -1.74609375], "student_probs": [0.027692176401615143, 0.8888246417045593, 0.040926385670900345, 0.04255670681595802], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a1359a4ab8400a9c733244de6eb962cba53ebfdbb2b5b2259ec79885868aead3:action", "state_id": "7011e8ba74458286794d13bd628f838e4d57440cc7aa2d7eec6877196592902b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.20703125, 1.3203125, -2.2109375, -1.65234375], "student_probs": [0.02647537738084793, 0.9010483622550964, 0.026372160762548447, 0.04610413685441017], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2631dd02a8494e3a07a53400d59b6e8f1103550b0bacaedf038cd2238b95b283:action", "state_id": "ae4813bbd01dd8cd7edd03d34f2c05a4affe3b734a2fd0f82d7c720ddc678794", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0546875, 0.91015625, -2.30078125, -1.58984375], "student_probs": [0.04392652586102486, 0.8518087863922119, 0.0343439057469368, 0.06992072612047195], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e6cac231b1dc6d53ca83ea49ae25cafd3e9462709a5903dc0c9ad7e5cd5c37e9:action", "state_id": "b92e332d082c957cc49314a5cf1c4e1d63b5f6d057c5cbff16eb8de9b084a31e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78515625, 0.45703125, -2.453125, -1.328125], "student_probs": [0.07996143400669098, 0.7527490854263306, 0.04100014641880989, 0.12628935277462006], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ce0312aad79da3bea4c2146d04e8f542dbc9c827611bdc5704b84cec39e7cc9a:action", "state_id": "3c12f20e6de9ee2f8d6766f268b68d07b72839ea2af0b8ee35d8179e01fdb252", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.88671875, -3.06640625, -2.8046875, -1.72265625], "student_probs": [0.34662139415740967, 0.1065426766872406, 0.1384160965681076, 0.40841981768608093], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b527834852ef22aacd6b56c9834ab2d5aa1fa4266a9a06f91ae1eae014b1c0c5:action", "state_id": "a972323c893f1b844d564ede730ebc94559a2b6e32e69fb5c6ad8375f05af8b8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.1015625, 0.93359375, -2.43359375, -2.44921875], "student_probs": [0.016280794516205788, 0.9207074642181396, 0.03175197169184685, 0.03125970438122749], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "94ee4bc43340a0bd1f54a53e199cf1ac4e0c41ba2af8c65df00129a9b91860ed:action", "state_id": "930d705afac645f323554eca3539587aabd97b3de7278382774abff6adf9adf0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.42578125, 0.60546875, -2.5546875, -1.51953125], "student_probs": [0.03987685963511467, 0.8263729810714722, 0.03505400940775871, 0.09869617968797684], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "94630271f1e8b101a7ed41422d0e835aa613d343d4c38d9dbe9a1999eea95b15:action", "state_id": "e697ad3fb784df27d9b17e7a269b94454babe9c704e7913462aec18ecee1581d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.91015625, 1.42578125, -1.95703125, -0.4111328125], "student_probs": [0.028955088928341866, 0.8137746453285217, 0.02762913890182972, 0.12964104115962982], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a456eb73ef3639edbfd56d2702fec1ae827ff474c7de47c8aa57436bd6b24d5e:action", "state_id": "d29596fce97b2e039bf5b6b7bf326c5b6d80f8397ace8ff426fe10e3050509c7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.078125, 1.20703125, -2.4609375, -1.609375], "student_probs": [0.03334100544452667, 0.8906435966491699, 0.02273659035563469, 0.05327877774834633], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "36d7ab1de6d2aa9c0a0cfb90e2b8df935d5f64432cf35c53011904945a7bf631:action", "state_id": "4f38e4aadff782335e7ba0c671e0c3c2a0e0e19a174fe91e41f42fa9c59182ab", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.125, 1.578125, -2.67578125, -1.62890625], "student_probs": [0.022834859788417816, 0.9264993071556091, 0.013164279982447624, 0.03750154376029968], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c231190ff0876c2c6aa33d7beda2f494e7d42e25648043032448c38892425ac4:action", "state_id": "90013adf1463e1b70fdac5b57d883787e116e16490b1b1f3396abbf544d94616", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.21875, 1.5859375, -1.83203125, -0.732421875], "student_probs": [0.05078767612576485, 0.8391095399856567, 0.02750512771308422, 0.08259770274162292], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "db5aec71f26f46df4a2dd2585f9f12d6449f276c12a11e65882018350d7ea8ed:action", "state_id": "f4ec7ad7919e7aa20a966e98211f4d85d5ba400c47d55fb9af3b87f8f55d1f96", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.81640625, 1.4765625, -1.98828125, -1.3671875], "student_probs": [0.03296864777803421, 0.8876041173934937, 0.027762385085225105, 0.05166475474834442], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ee0f0872c6fcfb511893b46966945787b51a145009fef0f41d10463d5f791fc9:action", "state_id": "2baf5ed56c891dd5fb899877a12f1b6b536123bd621d1eb65e9e2d836fc18f0f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.11328125, 1.5, -2.12890625, -1.3515625], "student_probs": [0.024263599887490273, 0.8998774290084839, 0.023887427523732185, 0.05197152867913246], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c11079146ba9c7ae0131b446cc9cc7f76d8d7b12a977b771834062e1d3b11c2e:action", "state_id": "890e9341a3e28c4c2d289caf6c0e5a15658dfd4031a7553ebf270ae9557855e6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2578125, 1.689453125, -1.671875, -0.671051025390625], "student_probs": [0.04441898316144943, 0.8463496565818787, 0.029359156265854836, 0.0798722431063652], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "30caf283b755a071ae95ad632fbe291e771ddd2f3a3877bd7d93cc55537747e3:action", "state_id": "16bb142f221a620b96bf871387b1931b715aa35110f9b81397be5f1ef9093b4e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6796875, 1.79296875, -1.53515625, -1.177734375], "student_probs": [0.02775491029024124, 0.894324779510498, 0.03207073733210564, 0.045849572867155075], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "451bbaddaf04cbe12755845010876a5fc91faccc5b3e6b9f35bc6b2e35ad630f:action", "state_id": "2cc65162ccc9e33494e542f6c178d1ad6f23aaffb48d60f5922342e690a4ade1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.01171875, 1.6328125, -1.8125, -1.5703125], "student_probs": [0.023786772042512894, 0.910196840763092, 0.029030539095401764, 0.036985866725444794], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dfe7f4d6354ea84fb0799b03b50e7e68f70fa9ea6ab0dcc4c1f4b7701b498459:action", "state_id": "30b0b9067e2f3a72f4bde9f1309b2df051042ec68bad89a9c92d66bf655cef36", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.01953125, 0.734375, -2.27734375, -1.65625], "student_probs": [0.05286911129951477, 0.8302488923072815, 0.040854085236787796, 0.07602791488170624], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5c285847d1ac4a097558da7b278e80d657c3580b83dae1ddc041dcc8da1bf8b2:action", "state_id": "5b48f66e50e8888ac93ab725b68b21db2d7d04d6e1fc9191f8f29cddfb474b66", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3203125, 1.046875, -2.4921875, -1.85546875], "student_probs": [0.030834969133138657, 0.8941172361373901, 0.025965647771954536, 0.049082037061452866], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4f67fe8d41ad2562bf30fdffdd6a3fe342db4aea94434fb75999a4025da1d24f:action", "state_id": "32757fad2f45043752e2a5a3edf1f1c88dd5b93532b137fe46f05587632aca31", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.703125, 1.02734375, -2.15234375, -1.55859375], "student_probs": [0.05514593794941902, 0.8459428548812866, 0.03519008681178093, 0.06372100859880447], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "65a852d499398944c476014fd68bdabcd3180c13f32cecfe9c388a6405c28c91:action", "state_id": "6b877b5574b0b56cd9e32a3cd46c8575e6218386322bb33aff1ffe6c4ce9feeb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.421875, -3.2265625, -3.234375, -1.99609375], "student_probs": [0.2922472059726715, 0.13070103526115417, 0.12968391180038452, 0.44736790657043457], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "63044a37214c7d2bb1bccf56d552a46cb5eaf1b74422cba0d9c601137f158b97:action", "state_id": "00362fa58fc2a400be5d9998e9a7356986ff3e7195e66a4fd04e84ab5c80dca1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78125, -0.75048828125, 0.849609375, -0.9296875], "student_probs": [0.04991962015628815, 0.1399346888065338, 0.6931687593460083, 0.11697691679000854], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "85b354473fba7538f2a7980b549bb028e5ec4f75745516a31e6d382094e4e11b:action", "state_id": "10cbe497e6024424709ba2ddae5f81af0cccac90ee77ebad8450c1c47e813514", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.7265625, 0.06640625, 1.345703125, 1.341796875], "student_probs": [0.1914171427488327, 0.09891875833272934, 0.35552504658699036, 0.3541390001773834], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4b9b406d632300d1ccc3c26b11be9590f82c56812cd87bb2f48bccb60e3110fe:action", "state_id": "466858f560198a85787e7ba1f88aadec0853833fa4b781456c76569a9c57697a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.12890625, -0.28515625, -2.62890625, -1.5859375], "student_probs": [0.10365020483732224, 0.6550894975662231, 0.06286703050136566, 0.17839328944683075], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cfb0032d48b95055c35a1c44e4eafac04db0a26b57d2d0ca1b4c966d309ce2ea:action", "state_id": "e7042aca95161a2fc2232146f21a09b8f5a5eb4e918008980c8dcebff0a9dac4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.33984375, -0.26953125, -2.6953125, -2.078125], "student_probs": [0.09151387214660645, 0.7254579067230225, 0.06413702666759491, 0.11889125406742096], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f20d6f4b4c0cca3c941e596c7ef2bbf988178f370d83cd9e97f46451adfe09d7:action", "state_id": "0af6bacd77c12c549b4b69a25a9069b16b313b11e35459250e1680f160efb369", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.60546875, -0.146484375, -2.6640625, -2.41015625], "student_probs": [0.06733231246471405, 0.7873119711875916, 0.06350041925907135, 0.08185526728630066], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "013477ad5c5df1ca0d304a246cf19f6c59aae27fc8844f5d2e610d64ce4cc75b:action", "state_id": "db3746547b4e24a9f6224ccd2e44ec975f6894efa71d8f6797dbaad521e723b1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.74609375, -0.265625, -2.46875, -2.4375], "student_probs": [0.06398774683475494, 0.7644528746604919, 0.08443950116634369, 0.08711990714073181], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4b39e7c95657e6ace262d6d1d4a03e0ff86a8454b86124b61dd76a349db454b7:action", "state_id": "7c17f3f234a51b8045de3604ee1c9d7cedf0aea152978c3d1d090d8425dd98a2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.66015625, 1.31640625, -0.0390625, -0.185546875], "student_probs": [0.08557073771953583, 0.6176400780677795, 0.15924392640590668, 0.13754522800445557], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8b47df8a6b5447c8ac4ea3c024e229e5690164413c0f9fd1d0427bee736bb9fc:action", "state_id": "f6da56de97743bc76bb9a85029c02524a396757a7413eb90bce6b0fefb83c2e1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1328125, -2.46875, -0.8486328125, -0.72900390625], "student_probs": [0.24455256760120392, 0.0642957016825676, 0.3249300718307495, 0.36622169613838196], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5e94edb28ec89956514be76b593bdce0ee3f5216ca99f536b72c9309ef0c6dd0:action", "state_id": "eac7053eb3dd3b3d2c13f43b903a5a816547dbbf4530df8f0a6cd002469ea681", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78515625, 1.17578125, -0.01953125, -0.59637451171875], "student_probs": [0.03396234288811684, 0.6560190916061401, 0.19851753115653992, 0.1115010678768158], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d7de982ceb4d34c6fb59ec8ebf95f6bb6198fb93d11d97b671ffc20491676135:action", "state_id": "7aad622b759c58983b12b647c7cbbf5f78349578044709a5f44d55b6862379b5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2109375, 0.6171875, -0.0390625, 0.24609375], "student_probs": [0.0678267553448677, 0.42203226685523987, 0.21894745528697968, 0.29119351506233215], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b1225828e6b5ba8ffb04f87ae6445d3f148aaa1540b9df6a98ff3b14bf62b146:action", "state_id": "8684f804a84b09237c54355c407d5c7ba243923af208410003a1b875a024361c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3515625, 1.380859375, -0.34765625, 0.15234375], "student_probs": [0.0423760786652565, 0.6513231992721558, 0.11564096808433533, 0.19065974652767181], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6b32ed5ac8dd4df295aa8ee4600b368e0f07b1c6c49c162f546e03beb8f1f224:action", "state_id": "0167256db4b7a89a729005a42f0b492c91f77e559c05abe2b77c284676ca7a83", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4140625, 1.736328125, -0.5029296875, -0.2109375], "student_probs": [0.033153388649225235, 0.7739719152450562, 0.08245707303285599, 0.11041764169931412], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5b42deea9d8761b0cc2c6e0ae0d5962c2324b39510de15e5543f0c1fc5b75a11:action", "state_id": "d5df17464342a125871b54ac4e66e27efdab86db75a224deb99ebd1ddb3bd254", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7109375, 1.814453125, -0.955078125, -0.77197265625], "student_probs": [0.025218257680535316, 0.8565895557403564, 0.05370078235864639, 0.06449147313833237], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e810ba1e956d7f16306ca1558e4d58abb56158f008819fdbf9aebfdb2afd4f26:action", "state_id": "9c902237fb4654462198c9d3ea775d665550f64257d60cba4258abcc65caa3ae", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.375, 2.12890625, -0.376953125, -0.08984375], "student_probs": [0.02464676834642887, 0.8193833231925964, 0.06686612963676453, 0.0891038030385971], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "61b006a9acaa977b7a477b1bf3d039450d07b3842db321cd6d596ca1d4b529a1:action", "state_id": "9b22895d404b485cce4990c4bb65de030ae8fd89a6f30fd1495c45f0e5d356be", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.12890625, 1.94140625, -0.2421875, 0.078125], "student_probs": [0.03531156852841377, 0.7609161138534546, 0.08570656180381775, 0.11806576699018478], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b86689c3a79c9c64881d37aacc44d7238893bbb1e59a0f6a09455b5c2259794a:action", "state_id": "e0c7b9c843d7807ded538acc0e9fb1d681e852f82fc0771284327948fc4993ca", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.86767578125, 2.04296875, -0.4736328125, 0.19921875], "student_probs": [0.042091190814971924, 0.7731577157974243, 0.06241973489522934, 0.12233131378889084], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b30befa4254a360542549e5dc4a40884162cb8a3c8d479d05776ff7900c06716:action", "state_id": "388de360ba691a47e248db8a10850eb6630518ef8ebb6b0b5da59d3d1db06ccc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.05859375, 1.71484375, -0.82177734375, -0.39453125], "student_probs": [0.04944751039147377, 0.7918319702148438, 0.06266029924154282, 0.0960601195693016], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "66b79d7fd181f13a1f2b1e535b67c76593d44285d13fb3fafc9ba54452e93e86:action", "state_id": "347d3772195519ecb7647911b9ae56f2e2b74cd7ccd198ecaf9546a7a5933c0e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9970703125, 1.779296875, -0.64404296875, -0.33203125], "student_probs": [0.04895120486617088, 0.7861842513084412, 0.06967568397521973, 0.09518887102603912], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "50ce6b2eb30bc1ae5511cb4ce93623a154cce8868c7c0ed6dc55b3d5ec1b08fa:action", "state_id": "8ca876f5adfde7ec7ab116bb908902e64dcf6e35675f34e8bd1171e43cde64c5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.701904296875, 1.974609375, -0.330078125, 0.109375], "student_probs": [0.051987212151288986, 0.7555994987487793, 0.07540125399827957, 0.11701207607984543], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3528b575ad8c5f59cb2c7e72d169686fe6681722cc375494e717582dfd36a269:action", "state_id": "806408460168b3f9129d35e0e23d1c388f5758c412e74ff197829449f1cd1097", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5068359375, 0.45703125, 0.33984375, 0.60546875], "student_probs": [0.11117205023765564, 0.2914726436138153, 0.25924113392829895, 0.3381141722202301], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b30841ab0391a60603afae0dd9361b36eaeb6f49f582b79dc53afcf024a95504:action", "state_id": "dbf223c3837e26ebbeb9f55d653ce8d0f4b0a384829e3c8577000907973f4569", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.08984375, 0.697265625, 0.93359375, 1.1796875], "student_probs": [0.10483318567276001, 0.23032231628894806, 0.29172390699386597, 0.37312057614326477], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "02d4567e8049be4af9c592eec7f37d080b3e5e0c61774235482af5bd1447ba5a:action", "state_id": "ca8c3830485979ed16bb187eec014c150b17b6efe1930bf5691754a0ce71e136", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.51416015625, -3.51953125, 0.00390625, -0.08984375], "student_probs": [0.2349158376455307, 0.01163311954587698, 0.3943716287612915, 0.3590794503688812], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6b421aab1aeee387f56d40060fa6686e40e9c77184c89a802d5677c9d9d23157:action", "state_id": "ecc31b1616526bdcf26742fdd59c70737a415299d2296aa100814183c700c090", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.09375, -0.3974609375, 0.923828125, 0.29296875], "student_probs": [0.16731633245944977, 0.12349186092615128, 0.46287837624549866, 0.2463134527206421], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ade85ddadc8be119b72237d974fd81f44a413e72ce4450b7bd1e9fa2d5a0532b:action", "state_id": "a1aa1065f42f4b1a324390152ab6e0038bf88d27b553727ce33fce017ed11b80", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7578125, 0.18359375, -2.12890625, -1.37109375], "student_probs": [0.09871020168066025, 0.6878663301467896, 0.06810799241065979, 0.14531546831130981], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "90aebc92b1f58d9359547614de4242134b847d4f6732e2695fcc530b7d429b97:action", "state_id": "602f26e69916098de49579bdada2a54eda712756b8b278f4a4ceb942714d2e6c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.88671875, -3.13671875, -2.74609375, -1.703125], "student_probs": [0.34347039461135864, 0.09840591251850128, 0.14543451368808746, 0.4126891493797302], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d18d86440f883ce45c72abd26660e2eac2f8af98bc037462937353485c69218b:action", "state_id": "e6104c95886ef6995d18445f4858b950a6fb124207046dbd15867324af621134", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9453125, 1.115234375, -0.3720703125, -0.87890625], "student_probs": [0.033259715884923935, 0.7097365856170654, 0.16038693487644196, 0.09661685675382614], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "441f9e3c4447f18697415df7e5b5ec15f1e361fe2468901dd27ffa37a1e2709f:action", "state_id": "b11f658660e5400d50a09af4fa1070505bb50bd9f26e9a431714a539385ecd2f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01171875, 0.697265625, 0.30078125, 0.50390625], "student_probs": [0.06760838627815247, 0.3734247088432312, 0.25119563937187195, 0.30777132511138916], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0960909d4ccd87d1d18945d860f75e34bd6050e6cbe04eff9f817adb35d67c4c:action", "state_id": "90fe22f59cac9923dde6b2fed641d5fe7a944755e0310dfd05e17b5dfe8c818a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.722412109375, 1.37890625, 0.52734375, 0.71484375], "student_probs": [0.059257280081510544, 0.4845433831214905, 0.20677773654460907, 0.24942156672477722], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "eddb739692e69f5e843f3b0e69a1d8d2370e001f17232d250ea37e235cdeaae4:action", "state_id": "e2acbbcba098e1ee7ac4e3f1fac2f45e9c0664399ac4e0d87b317f045ef451cb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9072265625, 1.81640625, 0.36328125, 0.43359375], "student_probs": [0.04233626648783684, 0.645017147064209, 0.15082977712154388, 0.16181673109531403], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "48a8d5dad0d9f5d252c33384f3a645ffadf91f49fdd73f613a2da65ef235def5:action", "state_id": "2f0b921e1570b742e6e23bf2d6a2109eb3023c389edc016f58521581c9f0d39c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3515625, -3.765625, -0.451171875, -0.605712890625], "student_probs": [0.17673316597938538, 0.015808986499905586, 0.43486329913139343, 0.3725945055484772], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "862429f2d179d095898b2bbb6ffd25f1b7fd5b38a64c02f3cf250b150c964d25:action", "state_id": "4ec2d425e8da34980d0660e11d71369bc93a2fa1396c35dc3c6f2363396ba05d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.56640625, 1.208984375, 0.16796875, -0.439453125], "student_probs": [0.038764920085668564, 0.6219789981842041, 0.21961823105812073, 0.11963780224323273], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "efa1f0da5bb2f84d0d4dde7ef393ee264590f9f68289f59ebb9e2a596de34604:action", "state_id": "ccd4a988145239c385016e49d7eeb040ad43957a94e8d88b1c28c7494dc77ace", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.958984375, 0.5859375, 0.41015625, 0.49609375], "student_probs": [0.07191971689462662, 0.33713123202323914, 0.28278616070747375, 0.3081628978252411], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3b8f3d06ae93146d06dfec2bb28e6668b1d44f5372e9de4266a7b24380eff4cd:action", "state_id": "7663b801b3d8b6150df81cc18fd8f95871ccc339fdc42084c47f39e3af6d3a09", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2578125, 1.27734375, -0.138671875, 0.21875], "student_probs": [0.04748677462339401, 0.5992072224617004, 0.1454150229692459, 0.20789095759391785], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9c85acd88c5c1320697c313a8e74c8ca1e2159be8372b291321fff2af600b4b3:action", "state_id": "b657f18970b675d4015615f463f98cf3d403f171cd6098832168ee750bab504c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.515625, 1.677734375, -0.5556640625, -0.375], "student_probs": [0.03214351087808609, 0.7833424806594849, 0.08394581824541092, 0.10056814551353455], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "71385afc163abe06d236bcc6e117322c99554f1a5b6d4f694f4ff4f9fe4f082d:action", "state_id": "f15da84ba7a5e7a348e5ac82596a90259e9900a0867afb0218716d48a864deab", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.94140625, 1.9375, -1.080078125, -1.234375], "student_probs": [0.018599271774291992, 0.8996706008911133, 0.04401148110628128, 0.03771861270070076], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cacebc05ee084e5aacaa814bd683732d99df0261f4271bf4f9f821bd2b756853:action", "state_id": "2565b79ed5a46d02f0c30ff93db4557689140def5585aca21f4e35be7ab69dbd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.890625, 2.033203125, -0.675048828125, -1.0703125], "student_probs": [0.017471155151724815, 0.8839312791824341, 0.05891686677932739, 0.03968065232038498], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6d299570c51757a985bed139ce92f4fc230e2a40a77d77b19fdbd45e50e825f5:action", "state_id": "d6f3f86da64dc46a8f02a648df8d680d6d9e2bbdf55da3671f982245112acf9a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.609375, 1.744140625, -0.6187744140625, -0.1875], "student_probs": [0.027441730722784996, 0.7849189639091492, 0.07389649748802185, 0.11374281346797943], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1676c945887f1241bda25b8d9c5a7de1b86b6b1abdc05f9c7ed925860130a289:action", "state_id": "5d1a9205e7d4746287de50ca8fdef5fde09cdb5dba7737973315eea58ee82710", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01953125, 1.931640625, -0.4736328125, 0.09765625], "student_probs": [0.040143292397260666, 0.7678751945495605, 0.06929368525743484, 0.12268778681755066], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e04e52c4917f8a204879512502815d3bc562b9c25417f93929ff7b219161e3a9:action", "state_id": "82e71515eb001d902db5e649f2b19bd6c1b6756df2607d08ca4eb0ec6658c2aa", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0087890625, 1.69921875, -0.7269287109375, -0.4140625], "student_probs": [0.05225345119833946, 0.7837685346603394, 0.06926684081554413, 0.09471122175455093], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1d063236182f90c88fb60cd8af5118fcb50c5333d97016a2f00b97b60be498ee:action", "state_id": "02fd2f9f082e63252ceb18d0210ebca2047a849287aa15bfefe5950cefaed60e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.064453125, 1.83203125, -0.62109375, -0.38671875], "student_probs": [0.044174086302518845, 0.8000087738037109, 0.06882023066282272, 0.08699692040681839], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a8e29f7308028f9b6216214b1b43844f5af4a0df957ea2f08b6767726b1ee3d6:action", "state_id": "294785bccba60c67869a139449392893085dce40000b1c7190457a14a7058856", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0244140625, 1.986328125, -0.4658203125, -0.22265625], "student_probs": [0.03955675661563873, 0.803099513053894, 0.06915360689163208, 0.08819005638360977], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5c4ee77eb9bf5d0456fc40922d7ca44487794329c94eb9f0b4aca2d50fe3d3be:action", "state_id": "6194bee4e58ee93242c381a33862a82a14d657e2c3e7e6cee8cb098715069885", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.208984375, 1.72265625, -0.3359375, -0.236328125], "student_probs": [0.04032658413052559, 0.7564614415168762, 0.09654969722032547, 0.10666223615407944], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e5f08eb973ec6c449f0c400c16e920ccc91700380459ac11d62399f3afee3ff0:action", "state_id": "3995536f0902484c311248f7cb221ab10adad3ba49de2717e84ee84676593cfc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.056640625, 1.94140625, -0.04296875, 0.03515625], "student_probs": [0.037338968366384506, 0.7485098838806152, 0.10289503633975983, 0.11125606298446655], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c8b09fffcff7ee8b836bd7d13fb5fde8eed2d31558dca7af93a4d95c54eca9c1:action", "state_id": "419baff4f6a7ef625ada2584ce9c642d0989fc9cdc228f9a91b884bd680afbe2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.052734375, 1.96484375, 0.2890625, 0.33203125], "student_probs": [0.03417456895112991, 0.6985871195793152, 0.13074888288974762, 0.13648946583271027], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "63ad3bb31bd43e42d11be786ba8db629afaee68850e8c6721bdaa68da7631470:action", "state_id": "953d4d0985bd7c5eec7c6417749c9289ad2b5bcf465d77ab201f4403f12ddcf5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9765625, 1.896484375, 0.36328125, 0.484375], "student_probs": [0.03728660196065903, 0.6596312522888184, 0.14237691462039948, 0.16070519387722015], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7b345c8c3512719d115e6879bb7d723e0966cd2e0c5c30c35cad2b62009291ce:action", "state_id": "69dc6d33b8f3251735f9f758e9113bf1457e5a19fa16a3461d7952af1396e099", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0625, 0.759765625, 1.390625, 1.36328125], "student_probs": [0.09565369039773941, 0.19209690392017365, 0.36099326610565186, 0.3512560725212097], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a2a566ca2612ae6da19abbac1a1f0c6c6f35755daa6e3a222e67c90393cb26a9:action", "state_id": "ce077f839fd98539df0b70ec95c8967fb974efdfeedca2d8de9bd601d987a0d4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.4296875, 0.625, 1.736328125, 1.7353515625], "student_probs": [0.1041712760925293, 0.12664008140563965, 0.384782075881958, 0.38440650701522827], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "59105cb55cf9b32474f6ee92998299d0f8ae37961e4a9f8f0d65ecbd13a0675d:action", "state_id": "7f69ebc535d634575c3d678548af34e9ca81e0674b4e355e9b47afabf17f1b4d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.283203125, -2.9296875, 0.28125, 0.234375], "student_probs": [0.2218601107597351, 0.015729891136288643, 0.3901378810405731, 0.37227216362953186], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5b987a735ad60594940c8c8303932a1d5a839e10c27913a6d5d6c634204caa2f:action", "state_id": "3f757bf95298106d85bc3535d851fd8670b97f7fda9d90b806261f98e4ea0ee7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.4375, -0.916015625, -3.35546875, -3.15234375], "student_probs": [0.06304169446229935, 0.7846836447715759, 0.06843110918998718, 0.08384354412555695], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "437a78ee17b318c733e298d0d4d9e6fe95e30920bf60b509beefbc35e52ec532:action", "state_id": "cce9081d729d42b42c3fdcb27fc048edfd760fc6d6bbd6fd69c706520fd2704c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1640625, 0.02734375, -2.75390625, -1.70703125], "student_probs": [0.08277063816785812, 0.7406140565872192, 0.04588919132947922, 0.1307261437177658], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bb814f5a5a1bc416404bf9795f232cfa35166d591aee41bde3834fde2bcbda69:action", "state_id": "c840d31a64eba84ca79f01f390ea4c2a444f0d7616671ebfc026852384aef261", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.24609375, 0.26171875, -2.66796875, -1.67578125], "student_probs": [0.0636833980679512, 0.7819074988365173, 0.04176459461450577, 0.11264446377754211], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5038bb120a9b52ced39f90f6ee301073e53671a0d669f3c6bd9a2e7142f1fe83:action", "state_id": "374cfcbfa5d7633c85cedeb70221ad22efa9ad8e8f67f636dbf5253ea950efb9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.15625, 0.64453125, -2.58203125, -1.53515625], "student_probs": [0.05007079988718033, 0.8240401744842529, 0.032709214836359024, 0.09317987412214279], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81eefe7719437b51450ad756cae47448bda32da0babdf9de7b2eefa5abbd3f48:action", "state_id": "8699ba7ede00c8d21e65a9b7f39c3236ea0f00b1cf70ee0574fdbe4eb176787d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.88671875, 1.68359375, -2.0, -1.2578125], "student_probs": [0.0254477821290493, 0.9041010141372681, 0.022722311317920685, 0.04772879183292389], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c875f6f9a4570e7d4192bc4048f4ed19a2b20fe6b2cdd64168bbdae65bcd65bd:action", "state_id": "e09c095d79815f31c9dfa67e088737ecca0d4df6e86059eee19b0e94a085c584", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7578125, 2.158203125, -2.015625, -1.28125], "student_probs": [0.018662530928850174, 0.9368596076965332, 0.014421286061406136, 0.0300565417855978], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4cc490badd136aa181291489fb43d40e70223db1c7b2c4480feb1412e4ce19ed:action", "state_id": "c62dffe0d60ca622af8546a2aa38645b935e0c0dfa716e9baaec47d25482d5dc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.66796875, 2.064453125, -1.41015625, -0.8466796875], "student_probs": [0.0215760450810194, 0.9014508724212646, 0.02792147547006607, 0.049051593989133835], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3c550489e9967fb6774fdd4e981ce51876e59869e7d94c69db82e38f43899bb6:action", "state_id": "6a0370266eeafcc761237bb6158313f8918ec2b39c5a0b6b83f9cfdcf0360e7d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9765625, 1.26171875, -1.67578125, -1.79296875], "student_probs": [0.03443251550197601, 0.8776804208755493, 0.04651535674929619, 0.04137161746621132], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "08e03072a2340d4616ae60eef7f2dff8eddb6160aef11fdc25c39c84aaa91b1c:action", "state_id": "843aa6b047861a6063029e4d4e3883b655ce16ff7802224d6c09712c9995e383", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.36328125, 1.283203125, -1.6171875, -0.85546875], "student_probs": [0.05700674280524254, 0.8040440082550049, 0.044223811477422714, 0.09472539275884628], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "da713da1c2eaab6867110b0d81e2c1adeb2afd748f21044350f2fda83f2461da:action", "state_id": "839a8bc79639f18230408744d80cded5f8352954544acd3d936db52d43267b12", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.640625, 1.18359375, -1.984375, -1.30859375], "student_probs": [0.05012360215187073, 0.844471275806427, 0.03554295375943184, 0.06986209750175476], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "82c999076e518147a9298e13b58084fee1cb0f538cca245f1b36fb1086c8cf44:action", "state_id": "639f4e7cee07143df041e4d33a701399c784760c6a164e0efe557a8376ab4f95", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.90234375, 1.52734375, -1.984375, -1.33203125], "student_probs": [0.028937645256519318, 0.8932182192802429, 0.02665860950946808, 0.0511854812502861], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "87139bb5d79598dec5ecb153127d6716b8abfb94ee2ca8ace26858d606f64a58:action", "state_id": "15d7d23eb3437dbfe5156ef0f292765b3c15846a262880013163da792bfd65b6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.09765625, 1.27734375, -1.53515625, -1.41796875], "student_probs": [0.029452824965119362, 0.8607377409934998, 0.05169131979346275, 0.05811811611056328], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3874f5543efe610ab5785abf5c853d033db00c6b551ed14b7cf224757085bceb:action", "state_id": "52ec88ac440625a2568713a0e2b9e0afd40050d8a6650a8be9baa9a5a018606e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.35546875, 1.578125, -1.9140625, -1.74609375], "student_probs": [0.01802307553589344, 0.9208034873008728, 0.028023939579725266, 0.03314950689673424], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "40d10315f2995d5ef4ac58f5fab21d91babe6ff04e96ac1bde5fd1f550a39be0:action", "state_id": "0573fddb69fa72195f2f756e8061440f303896e3a5390583d80383ed33affad6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.23828125, 1.6953125, -2.078125, -1.73046875], "student_probs": [0.01820644550025463, 0.9301719665527344, 0.021368801593780518, 0.030252786353230476], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "38455fa47c6059b046e165575f453dd933bc0861821057272140667a32afaea9:action", "state_id": "afe1184a23b79058c4444ff53af7a2a123e3d52d19042226cf4a0ecd640629a3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.28515625, 1.62890625, -1.99609375, -1.72265625], "student_probs": [0.018452804535627365, 0.9245238304138184, 0.024637725204229355, 0.032385680824518204], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2b00d99a398cb8905483debcbd4d2e4f5c269baf9e69366b7767a0a874f0b912:action", "state_id": "56fcbd1b8bb96f20624b506abf04b02a898f9a93ffb70b4fd545c86b396ea3d3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.83984375, 1.658203125, -1.3828125, -0.94140625], "student_probs": [0.026256384328007698, 0.867795467376709, 0.04146876186132431, 0.06447945535182953], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9890e2ec105da5a21f53adfdcc709a544fdaa787464d1d0b6c072b199c483c37:action", "state_id": "a67c909e58911f3dafb6fb9bab882785a14378e02767fb9613d5df069c16bb35", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.20703125, 1.625, -1.75390625, -0.96484375], "student_probs": [0.01915980502963066, 0.8843437433242798, 0.030142605304718018, 0.06635387241840363], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "35430e8e0e350bb59d86320a6170a21eb6d62350a6d5b8ff2c3318189fc8d67c:action", "state_id": "e516e992d4dd6ceb2a53d546320646fab15db520cfc181f36970f1b6b095d6d4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.03125, 1.62109375, -1.703125, -1.10546875], "student_probs": [0.02300058864057064, 0.8870164752006531, 0.03193315491080284, 0.05804978683590889], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "62af459ddcbc02885956df24fb8032fc199a050333f5ba7e3574e1216cc9eeca:action", "state_id": "f9e8a12d4b16bde039c8f929733c74961f4f69186b552a88c99fc6285c494b23", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.10546875, 0.421875, -2.26171875, -1.52734375], "student_probs": [0.061887916177511215, 0.7748494148254395, 0.05293554067611694, 0.11032714694738388], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "63abaeaf0074e8c3000d2633f6d5db69e73e95b7b7f94c020d2a9c1cd6038ec8:action", "state_id": "34c7dd69d971c1f2cf7473650f5489d83ecfa85894c9958e0f884bc6a386ca0a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.62109375, 0.80859375, -2.05859375, -1.2734375], "student_probs": [0.06936387717723846, 0.7876498699188232, 0.04478468373417854, 0.09820158779621124], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "72bfd2e133f0ccca510a2f47cc8592d497b1db1ed12856b1db16f526741d85f4:action", "state_id": "1eae323828d1bd667f3fee9815236547738c723b60ef756d8305fc0883ed591f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4609375, -3.11328125, -2.5703125, -1.4921875], "student_probs": [0.40150994062423706, 0.07692943513393402, 0.1324039250612259, 0.3891567587852478], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9889b27efd0f9730004c74177b1150a4003badf7b4bdaf66b43f1663fc7d12d1:action", "state_id": "dcab839b896e71ad2ac27137204c984df3a9255cff3672994daab0fbe3dd55d6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.5, 1.12109375, -1.37109375, -1.59765625], "student_probs": [0.022760337218642235, 0.8507456183433533, 0.07038116455078125, 0.056112758815288544], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c9b9fbe747662362b6e80230323ed63b3591e40f7a8c79fe2b262f7bd114e228:action", "state_id": "fbbcf6c07a0ca681914c03dd946907f5edac064230ee70d51c8110adee195b1a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1875, 0.57421875, -2.07421875, -1.390625], "student_probs": [0.049589481204748154, 0.7848538756370544, 0.055537592619657516, 0.11001908034086227], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9596b937864434496832b15717ed1006fc5bc53fd0d8befe40569942d4303363:action", "state_id": "ca554f8195e5be801e80845c2b42c0b64fd611460e69adffaec78805ee3f05fe", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.9921875, -3.31640625, -3.5546875, -2.5859375], "student_probs": [0.2635704278945923, 0.1905856728553772, 0.15017789602279663, 0.39566606283187866], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "909d43ab4a2f5fcb20c7a5b82d4aac65a9944200ac5fa531408699925ea034e4:action", "state_id": "ad8eb7a41a50f690e93cda2572f83fc44409dcb9a82ff4645aaf6d9e70ca5e76", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.7109375, 1.046875, -1.86328125, -2.0390625], "student_probs": [0.0207698754966259, 0.8900842666625977, 0.04848041385412216, 0.04066544398665428], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "02666a87bb7943c0e9ba62e7265234b259e03056c97bc358635e7052b9027b87:action", "state_id": "eda50e9fe9755d677c887c1cb03f1fda6523a88b2cf100e5e0573b97edf9f7c3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.298828125, 0.26171875, -1.001953125, 0.203125], "student_probs": [0.08622531592845917, 0.41055533289909363, 0.11602885276079178, 0.3871905505657196], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4ff33a1327291487b7b7e0fe96ce1848dedb39c204e1176781fa7660dc0eff90:action", "state_id": "141e1899fe1715da658d75bc21bb345ca892c410c2590586afe666d9b89fe6f2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.24609375, 1.03125, -1.96484375, -1.203125], "student_probs": [0.031577929854393005, 0.8369817137718201, 0.04183395951986313, 0.08960644155740738], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "df13a5c803a25cd864411bf36dbbab7c06dbecaa193eb562afc577326823240d:action", "state_id": "7dd5f55f08eb12caaf5264fa206ea5cf688293327df38e288414d38be099cefb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.89453125, 1.19140625, -1.61328125, -0.994140625], "student_probs": [0.037490636110305786, 0.8205941915512085, 0.04966702312231064, 0.0922481119632721], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "33ac78e6249968fd53c71617882f5943f5a77dcf309f00bc6984e160b4b2c9b4:action", "state_id": "646e293b345feb8d07fb6303e6278ade6da273e2110a65ffab3f67ecc9744220", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9453125, 1.71875, -2.140625, -1.27734375], "student_probs": [0.02336864173412323, 0.9118335843086243, 0.019222520291805267, 0.04557520151138306], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "50b9627f002fff850ee4842ba8eb49efa7c6bba20bd80b0678a48823ec199333:action", "state_id": "b28e44ec32c7564ae8121d1d0dcf6e5cfb69e40764f570360468e9d3e62ca3de", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2734375, -2.9375, -2.8828125, -2.19921875], "student_probs": [0.318929523229599, 0.16417084634304047, 0.1733989715576172, 0.34350061416625977], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3ec24725f232b07427bfa2a8a45be66b779e6a28f2388934d3ed83833373352a:action", "state_id": "c082aca9328fe7a743005568f3f3780cc16e744722fee1abc07b8fec5bad89fb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.85546875, 0.8203125, -1.8359375, -2.390625], "student_probs": [0.02229994907975197, 0.8803905844688416, 0.061813123524188995, 0.035496290773153305], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "538c85a32b373b4ea2e0224fecb793cbdbb3adc34c598de77490486216589063:action", "state_id": "02591abb6873e0738a4447cff894fba30ad1a46a67416bef8f6cac8593257136", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6640625, 0.375, -1.41015625, -0.9755859375], "student_probs": [0.08359014242887497, 0.6422566771507263, 0.10775195062160492, 0.16640125215053558], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "83d79307793d60dc869a5dd27a638974c9dfe3f2d4c168ea7f64752ad0717fcb:action", "state_id": "7640d8ca44bfa421b36942c32e527ac30267422201672efc9cbcc1f69bc05646", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.921875, 1.30859375, -1.23828125, -0.642578125], "student_probs": [0.03138080984354019, 0.7936680912971497, 0.06216489151120186, 0.11278614401817322], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3aedfdc8f1d1884483d164fd4b815c562861bde53edc62bc9eb1c29fab79e5ee:action", "state_id": "13474ecb9159d20d7362fb2e3dde2534ec3253172c01565545973a0408ef3f58", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9921875, 1.5234375, -1.46875, -0.7919921875], "student_probs": [0.025223523378372192, 0.8484422564506531, 0.0425727553665638, 0.08376140147447586], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "846f1eefb22c79422c0b9c80411c01ea286d85dc05a759124422a05aa779cbe9:action", "state_id": "f6cf8097539462bf69354c6f29030805812b267fc93a787ca5e7f242029165d8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.25390625, 1.6015625, -2.05859375, -1.625], "student_probs": [0.019477227702736855, 0.920313835144043, 0.023678287863731384, 0.03653067350387573], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3b9e0fe1aceed18ac7e45dfcb25b62386fa864e0627797ec2217e00f2ed21408:action", "state_id": "6be16996231957f3c2710625f20d20282f657310c0ae07da3e2354674f4de647", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.96875, 1.939453125, -2.015625, -1.23046875], "student_probs": [0.018568063154816628, 0.9248635768890381, 0.017717769369482994, 0.038850631564855576], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3cfd9f2aa1f91e917c00de78330ced3b27296c245f59aad1819a87ba46203a2b:action", "state_id": "eb24aff61f6986792f218e4abce5d6d35c25468b7dcdac3b57fe73ce94aede3b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.109375, 1.703125, -2.12109375, -1.4453125], "student_probs": [0.020327487960457802, 0.9200923442840576, 0.02009066380560398, 0.039489567279815674], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3c36a6186f996f2c1c5a554d395b256cf4d99d559f3afcc0125d209b2a494807:action", "state_id": "4424400f485984ca68e2284ebaf94bfc66e6f206de26dd11cc02865037d0b682", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.046875, 1.4375, -2.55859375, -1.76171875], "student_probs": [0.02814405970275402, 0.9175539016723633, 0.016871361061930656, 0.03743075206875801], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0d3824511ba4b1bfce618bc4a55282ab4efeb9adb19506fd59ebfc17df3e57d0:action", "state_id": "640860a681c415c8a4ce0fc9a08bf3bf1a93f22ba6fbeee7405e2f92efaa156c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4375, 1.2890625, -1.90625, -1.33984375], "student_probs": [0.055528901517391205, 0.848496675491333, 0.0347491018474102, 0.06122526153922081], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1208d8259cc95f572616370858b8c86108742119a656a5b3e28f0613366096df:action", "state_id": "baf5f4642be46d71a69ba256c1791bc0f5c3f0a6f70dde3dfa86c8d975006a81", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.28125, 0.94140625, -2.640625, -2.05078125], "student_probs": [0.03564809262752533, 0.8945778012275696, 0.02488638088107109, 0.044887725263834], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "29e45a2b770768d08ff74cde32200e03610fe2dc06e45e2934e72aab197eed11:action", "state_id": "58cd3fc0cd807287a7442dbe5d12386606389b40b12dc5255f8f84cd7f6229fe", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.09375, 0.63671875, -2.70703125, -1.6875], "student_probs": [0.05439860373735428, 0.8344786763191223, 0.02946070209145546, 0.08166196942329407], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c7fa7c790fc18c67634e6b18b8e94de3e054c2178989fc7ea2529399879e00bb:action", "state_id": "80abe17746ccc417d1fccbddaa025dd880afe520de2f8da9a8e1b6cb84c828aa", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.32421875, 0.55859375, -2.734375, -2.0390625], "student_probs": [0.047943320125341415, 0.8564808368682861, 0.031812623143196106, 0.06376317143440247], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "820a76638cf7d96b9836f876a32243722e08097e86abc2aacf368574e87d247a:action", "state_id": "cc3c594e651dd123c8eefb75ac8ca2b9f6ed59475be4a754e97b85c836857a4f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.27734375, 0.9453125, -2.0703125, -0.916015625], "student_probs": [0.0825112909078598, 0.7617294192314148, 0.0373363122344017, 0.11842304468154907], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "82cd76e637773c13755a86e03cfb6d8810268b9ab35a3849ac3655bbe6f4c7b6:action", "state_id": "06cb4921090cee0ba70ea3ba3bbb38f2022a15d98dbd85d16afca80d973abe43", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3828125, -2.09765625, -1.74609375, -1.107421875], "student_probs": [0.2855752408504486, 0.13972297310829163, 0.1985863894224167, 0.37611544132232666], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f48e522ca95c524076e89faa071d5aa8b8747229b8a7b56ae65d0210b706f021:action", "state_id": "39709e1019b68222952690c711996744bc269c418a93fc986c325bd9d138b5d6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.41796875, 1.201171875, 0.25, -0.52685546875], "student_probs": [0.04451734572649002, 0.6109526753425598, 0.23600372672080994, 0.10852625966072083], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7bc1bb43f16aef88723071ad31f1ebf1b8f7809d2193c7391a044332c15421a6:action", "state_id": "96f262539b28bf2419733be1ccfe07dc73747168d7a66c11cc2a4eb604cc31f4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6064453125, 0.5234375, 0.49609375, 0.30859375], "student_probs": [0.10412361472845078, 0.32229316234588623, 0.3135998547077179, 0.2599833905696869], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e18b3fa09fadad774c10a9519e9d2c049fc0cbe07a954b42e375c44d8b93d150:action", "state_id": "dfb43a393b1a85a55716798a515c8f162932f520fe72dba71fc63aa2d97eb40f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5810546875, 1.3125, 0.29296875, 0.53125], "student_probs": [0.07644771784543991, 0.5078376531600952, 0.18320955336093903, 0.23250500857830048], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2df35c5bbe85ad2142361a474f26de460bc64ed2408f0d2b10fadf4ed219953e:action", "state_id": "3cc4f4280150cc4807197230a8bfe36932dbad276168c31f2d1459e766e19b3b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.09765625, 0.83203125, 1.095703125, 1.001953125], "student_probs": [0.12095772475004196, 0.25209754705429077, 0.32815563678741455, 0.2987891137599945], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "256543a5a401b3c297d25548e873697e41701a40e475c74479e8567eb4c775bb:action", "state_id": "4c3b51d066dbbc7ca851cddd3a3d5b2670b299ae78e70e2e21adb43d7a8bae31", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.627197265625, -2.6171875, 0.015625, -0.03125], "student_probs": [0.20604592561721802, 0.02816580981016159, 0.391866534948349, 0.373921662569046], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b437026b51be7b897c9de01ee8bb5bac178bb23eeb455a0a253bcb8404faf29e:action", "state_id": "ecdce9688964f3af613e5758d24286f0b11c43bfe9683c37058213d07e16d786", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3828125, -0.6656494140625, 0.51171875, -0.615478515625], "student_probs": [0.08437351882457733, 0.17284870147705078, 0.5610358715057373, 0.18174190819263458], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0ad9b6768423637d45b8c624cb5d84a1f34f116af4fc77ef6d551027bddebd2f:action", "state_id": "0921bafe55fc8c42994923e07f4bc2b06e89a55a124de12e9584490077083310", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.88427734375, 0.33984375, 0.34375, 0.3046875], "student_probs": [0.09009542316198349, 0.3064303398132324, 0.30762967467308044, 0.29584455490112305], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f4507a103c71c12a5dac03716377dc09c8d3048435a18159abd1d1f621cc592e:action", "state_id": "c5d9898eec7167888aa7ae2e71e2fdf863e227e8211e486429248ffc80d5e949", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6068115234375, -2.80078125, 0.01171875, -0.0390625], "student_probs": [0.21132881939411163, 0.023557530716061592, 0.3922681212425232, 0.3728455603122711], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8414f50f31efd880015f7105a586c1522a1a3969bc7bc33adce83fa09589163f:action", "state_id": "3b853e303f0ec3312969f816e4a36cdf2e0b246484071c5dcec305961688e644", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.0, 0.23046875, -2.8359375, -2.7578125], "student_probs": [0.034790072590112686, 0.8798934817314148, 0.040992721915245056, 0.04432370141148567], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "07d25846836f6fa24667b409736d7f0a70d28e364b839e91619d183ee3c56353:action", "state_id": "a12c91c5803bffd1928fbff94565f66de10e1e7326f635045c933f68cd83da1c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.609375, -0.0078125, -2.30078125, -1.140625], "student_probs": [0.12407496571540833, 0.6155083179473877, 0.06214558333158493, 0.1982712298631668], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a976ea63c9dc5760552ee65e3e23f1ad31a793b1dc10bbfa3b4a4087bef810a0:action", "state_id": "1b4ebef496f8a6d7188806a0c14c8cc4715023a8ae0fabfd203d600dc1037da0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.015625, -3.7265625, -3.17578125, -1.9609375], "student_probs": [0.39210397005081177, 0.07085174322128296, 0.12289997935295105, 0.4141443371772766], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e6381bb7e317992b5d1cd859138aa29cc28126c4452c822ce6d11e77dc474388:action", "state_id": "ee3a361e5629832cd3702e6e26d7e4b8a016649d96c2850190336f25ef5dca07", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5625, 1.2265625, 0.20703125, -0.40625], "student_probs": [0.03800567612051964, 0.6181913614273071, 0.22302120923995972, 0.1207817941904068], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a10fbb4ddbbb453ec0f6f0de6ee93ab4bf7d5a0dee9cce19bb9aa1d37cc1b177:action", "state_id": "3e87d38159cdce2483f88ee41373d3f35d4601741c9c34449d7328d77d1a2c74", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.09375, 0.55078125, 0.20703125, 0.34375], "student_probs": [0.07111918181180954, 0.3682965040206909, 0.26116132736206055, 0.299422949552536], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9eb9b06a701664bd6a7053c723886791947acbbd9887de402f3cde8aefb293df:action", "state_id": "ba61549a732895334da5399d61aa10186f3eb3e00a7491ed38d5584c3568de31", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.28125, 1.36328125, -0.197265625, 0.2109375], "student_probs": [0.04448380693793297, 0.6261916160583496, 0.13151350617408752, 0.19781112670898438], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "afb5fabe9e93b34b4161f84ca0143fa2f800b1c08b469003320a6ef9e5e95df0:action", "state_id": "3d381d1f53d8fb6efdbea0e30880ecd15f525c9921fe44066e738b57f243a806", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5546875, 1.720703125, -0.64349365234375, -0.4150390625], "student_probs": [0.0302420724183321, 0.8000103831291199, 0.07522080838680267, 0.09452671557664871], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "29d077fcdf8c9deda836ab689cc931ae3a7573acbeaac2903828b9dcdeb5b437:action", "state_id": "809a344e6b3ffbd50c6996c75dde335ce7e00b3f45232f60335ce76c2b269e91", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.94140625, 1.904296875, -0.8701171875, -1.080078125], "student_probs": [0.018840547651052475, 0.8815788626670837, 0.05499819666147232, 0.04458241164684296], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2fc7a479f75fe57fdb8b864f66bc8f029199e7adc49ccc446203434b6add1356:action", "state_id": "3f66aa981186537eb4e992ec7cf243216b7cae4706186d26341b41816b87db22", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58203125, 2.087890625, -0.439453125, -0.240234375], "student_probs": [0.02118213288486004, 0.8313741683959961, 0.06640259921550751, 0.08104097843170166], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8ef9f326859f1c8124d310c351188531cfc1ade0446fcf6c98136701222889fd:action", "state_id": "e2c87829884c164ebd2b5b449d58b647b5b519adf19a22c1b1cd0bb4e464828e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.32421875, 1.875, -0.4853515625, 0.046875], "student_probs": [0.03147943690419197, 0.7716670632362366, 0.07283537089824677, 0.12401818484067917], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a2d1ed5b1fcd219c6afec8fbbc8e3dcf48428181eea855acb124bb0ed373a616:action", "state_id": "7582fe740ed0245f7f1b18603ca88257f264f605860fc01a0ab84717c11ffc83", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1484375, 1.986328125, -0.5419921875, 0.03515625], "student_probs": [0.03438406437635422, 0.790257453918457, 0.06305696070194244, 0.11230146884918213], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "762b64b81e71aee9d9dc96c0710fb407c6a49d87639009a1271f3781b88326a5:action", "state_id": "8353912fb4c944326d74aca17a9312a61b1137e263c96e623189a6ac8e60ff4c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.17578125, 1.6953125, -0.80712890625, -0.4482421875], "student_probs": [0.04510175809264183, 0.7963310480117798, 0.06520743668079376, 0.09335974603891373], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "11cd8526ef6b0972aad082dac3b36a3e039e36cef29ff9446467d4fb31025ad9:action", "state_id": "a5feffcbe1d52e24caca8cfafe828b8ffc864e8a488c04e54523c70e248d2998", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.11328125, 1.8125, -0.677001953125, -0.388671875], "student_probs": [0.04299289733171463, 0.801765501499176, 0.06650746613740921, 0.08873409777879715], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8f2b5670d61ef01f205380f6e21f9a81149fb0cdd54d33ff91dc1fb5e8388596:action", "state_id": "d13397eb7e8658935d194f6bad2efe14e50ee19b5872b6cb0546b9a613a58794", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03125, 1.958984375, -0.5029296875, -0.08984375], "student_probs": [0.039761416614055634, 0.7908682227134705, 0.06743858754634857, 0.10193172097206116], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "530f8cba996817a896ce6637f8dfa157bbd957ec8eaa7c257e527c7a834f48ac:action", "state_id": "0badea6838aee914cca7dc05a98813f7bfccc4bf1f11db65087dc9bd3c16c11f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.236328125, 1.671875, -0.341796875, -0.03515625], "student_probs": [0.03985009714961052, 0.7302069664001465, 0.09748086333274841, 0.13246212899684906], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "50842c691e81679bf89cda5544a47603dcb493cdc4597a375d3d148d7d8976b9:action", "state_id": "2869277501310949333c8716a8414392c3efb18cd1334c024a2690f9682b6528", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.994140625, 1.927734375, 0.04296875, 0.15234375], "student_probs": [0.0391477607190609, 0.727212131023407, 0.11043781787157059, 0.12320228666067123], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fcddafd8703462e218dba37c0b391ad0995840be704b4f9a6a5ea32e64a3f12f:action", "state_id": "da661a8d80a8c03fc6c1be57e37b56d9ff63719e271320918ceac0a2bb02abe7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.80712890625, 2.013671875, 0.546875, 0.5859375], "student_probs": [0.038925085216760635, 0.6535634994506836, 0.15075302124023438, 0.15675833821296692], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "be88c3d3df04c1974770f08ae790f5b20691a0beef1b626ee1d2d4fc3e92edb8:action", "state_id": "5697d431d5aa8b4686e0000cfad0408ce91f395cf078bc84268e8b2ec0d70f05", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.79931640625, 1.970703125, 0.54296875, 0.638671875], "student_probs": [0.04000169783830643, 0.6383848786354065, 0.15311771631240845, 0.16849566996097565], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f38fd9d37b7507853e064b62811cee1012a03f0090944b23cc7c4d73c32ba4ab:action", "state_id": "bbec95b9a3d6d3e4eac76f5ded57191d2f6ef09f76923aa7630d9d883863cd6b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.09375, 0.849609375, 1.314453125, 1.302734375], "student_probs": [0.10132645070552826, 0.21576866507530212, 0.3434531092643738, 0.33945176005363464], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8bf44992dec988f758a9371ec6a67a0deb082dd0b41b07a687fef26cef788b52:action", "state_id": "446eb8a665d31a9a75438c5d7031be4b1374827a9b37fb9b8c54fda0eda45bf7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.46875, 0.75390625, 1.888671875, 1.73828125], "student_probs": [0.09974116086959839, 0.13265274465084076, 0.4126089811325073, 0.3549971580505371], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1378de18221da8c21df72a956debc9831b75717073a018638a891fa82d12b2ac:action", "state_id": "9f5a3d61f1411140ded45d346ce4e9a2c071dc2c6630e4efeeb750c152e2e17d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3720703125, -2.8203125, 0.20703125, 0.15625], "student_probs": [0.21896398067474365, 0.01892843097448349, 0.39072689414024353, 0.37138065695762634], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "61a6ca839035374968ffbcc393966134c1e646f0bd72facf9392fd5db7faeef8:action", "state_id": "015462597ff34a58c4e292a808bf881b596e33707f197408604e76ae2fa8aab6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7265625, 1.19921875, 0.01171875, -0.537109375], "student_probs": [0.03493860736489296, 0.6515627503395081, 0.1987154185771942, 0.11478324979543686], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ab2af1e31c4a5ad7ca0f7b0c3e816abbbe99e276210fde145abda7f29431f155:action", "state_id": "3e0159fe5a095068e4d7a4ef497c06c7f7556769db4bf7dfb9952bf3e45ae547", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.23046875, 0.7109375, -0.123046875, 0.21875], "student_probs": [0.06555282324552536, 0.4568076729774475, 0.19839859008789062, 0.2792409062385559], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "de1573ba2d8f242c5f21f8e20c70a9bc9edb6d015fe032caaac5123a01afc5ec:action", "state_id": "264663f4173ed36f32e6b59822fada478553526cf631f19670a1e291ca369eed", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.45703125, 1.427734375, -0.4326171875, -0.01953125], "student_probs": [0.03861750289797783, 0.6912290453910828, 0.10756761580705643, 0.16258575022220612], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f8ae905e6093926312a6a91afc63ff48de6d17281f1375747467f7a56a84774b:action", "state_id": "7b06adbc86b41c4aa1feb1bc754ad57d67ff4cd9ab1083f17153dcc7055a2e43", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.67578125, 1.888671875, -0.67822265625, -0.56689453125], "student_probs": [0.02377399429678917, 0.8397005796432495, 0.06446682661771774, 0.0720585510134697], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "15cb7b02c8abb0f15754c214aed212cc352cf3c1922615f9d618030355bfc626:action", "state_id": "44391715c232dc1cf09deba7a8fd5fe2beaf1506b65fb292c44a84876e60e583", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.98046875, 1.900390625, -0.7978515625, -1.27734375], "student_probs": [0.018265238031744957, 0.8852401971817017, 0.0595976896584034, 0.03689679130911827], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "871daea840f3d2cbb77d681b18e899ed6f55ca4ea4f874778be7048f71eb1d69:action", "state_id": "c58affb0f424eeea343b6c05dbc7e1b12757755f04735b384a1022ad75a8c687", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.49609375, 2.01171875, -0.56689453125, 0.09375], "student_probs": [0.02391735278069973, 0.7982459664344788, 0.06057022884488106, 0.11726637184619904], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cb367a9c05a01768cccc21b996204db8d72bc8e6d770b0ebe162b77ecfef1a07:action", "state_id": "638106610023be2fb1dfc8e034a6ad1888776e35618effea1a8bb02f11f63c00", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4296875, 1.828125, -0.649169921875, -0.00390625], "student_probs": [0.029997309669852257, 0.7797085046768188, 0.06547217071056366, 0.12482202053070068], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0e3c8d888d9997f330790cfb440ff9db71a876bf8baf4596dededd447a855ec4:action", "state_id": "1df0bf52baf4584e63d34b7efc6dafcbbfa179cc4ba424a56b24d625bd97b691", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.15234375, 2.021484375, -0.57421875, 0.03125], "student_probs": [0.03339167684316635, 0.7980209589004517, 0.05952710285782814, 0.10906025022268295], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0180a15a5c861abeec126a12ea1efda8518abafc4a08fecc708b0df37a3fe78b:action", "state_id": "2f42812082a0edaad406a91c50431029b79484627d97d5aa06cdce722da2d3b1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.091796875, 1.7109375, -0.82373046875, -0.41015625], "student_probs": [0.04813656583428383, 0.7937563061714172, 0.06293538212776184, 0.09517171233892441], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7614f804a3814f369358ac4b4c9acb18a51336a26cd2bc1f0d8a72be4c749552:action", "state_id": "b1938b067762761033269f0328a702fbfdb1d77872f3a79bb2ccd2466ad22cc6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0859375, 1.7890625, -0.66943359375, -0.359375], "student_probs": [0.04482287913560867, 0.7945045828819275, 0.06798061728477478, 0.09269192069768906], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e747530c7b0d5d60b3d3b181b90ed911255eeb65c7fbdae5388f78c9ae09468c:action", "state_id": "80e480139f59896c2b0803b36455210a4fbb798f021d7e40b691026b5a3a0d97", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9853515625, 1.943359375, -0.5302734375, -0.142578125], "student_probs": [0.04236820712685585, 0.7924339771270752, 0.06678485870361328, 0.0984128788113594], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8ed68d549ba58214e136979fd86739b0371eec1d3f5fa6cdf30e8612290d8fa6:action", "state_id": "af7ea9e5ba4784c606b4e8483cba6dbf8908f9fc136a126e5c110255bb311531", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.08984375, 1.65234375, -0.349609375, -0.23046875], "student_probs": [0.04766668379306793, 0.7398298382759094, 0.09992972016334534, 0.11257366091012955], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5053c6985d348f919fc797711b21088db473b5089e903649ba844d10eacfcd22:action", "state_id": "0b5209132a887d9bd0542f3f40a1d9b7fd637d05e449cd82a11e79900932c930", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.87841796875, 1.861328125, -0.0703125, 0.0546875], "student_probs": [0.04701656848192215, 0.7279600501060486, 0.10548888146877289, 0.11953455954790115], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1c8639951594349ac65a27e5670dda52e88a1ff4598596bf76bda7458c879c43:action", "state_id": "124f644f79a0df20b44153a7720df0d41a364444a0fd68d914a4f052c409f542", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.98046875, 1.908203125, 0.2578125, 0.23046875], "student_probs": [0.03879617527127266, 0.6971451044082642, 0.1338343620300293, 0.13022440671920776], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5bd64b0c83027884d1394680eb372a12d5568346aa95da8214d84843ee06f122:action", "state_id": "3d4ce4c6c323f0d78e767f258bb555807d77b8b0c346652bd162c5d9e8b00f8d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.095703125, 1.810546875, 0.28125, 0.21875], "student_probs": [0.03707326948642731, 0.6779993176460266, 0.1469143033027649, 0.13801321387290955], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "98405f267b10b53c9423fdf25db4642c8015738c100e13ef83b0c8b03b822219:action", "state_id": "962d701b467a38eb6448db6c727828679ac9b02529ca5e3c621569d2c7831625", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.345703125, 2.0400390625, 0.75390625, 0.8828125], "student_probs": [0.05468583106994629, 0.5942777395248413, 0.1642211377620697, 0.1868152618408203], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a1244ac69b157374ab45b5eed8b73208bb5a04599fdf9c053049974561217a03:action", "state_id": "3c6e2d011fa1342e72abf0b610bad0f670922992b681763d97ea2801da0e5f7e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.921875, -3.09765625, -0.16015625, -0.228515625], "student_probs": [0.19026242196559906, 0.021598482504487038, 0.4075334668159485, 0.3806056082248688], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3b929eeb97c720054768d956476f79e609bc9800bf3134f8ea23f47996861cb5:action", "state_id": "c137e15bdb57d472fa19a068b5accd31cb05fed2dd2012836d9e0e614ca2e1f7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.01171875, 0.265625, -2.86328125, -2.8203125], "student_probs": [0.03347140550613403, 0.887168824672699, 0.03882751241326332, 0.040532246232032776], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "65427503cba714c2f31217868a1ea8040cd4c7914e6e5db519947523bff38d99:action", "state_id": "7808d1e24c8cb41088d3f5a9aab350a3b783c952351fa13dc4ac3063e6bb6648", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.23046875, 0.38671875, -2.3359375, -1.76953125], "student_probs": [0.05819838121533394, 0.7971516847610474, 0.052372872829437256, 0.09227700531482697], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ac2a093cf16df140e44a032f66cdfa96a73fa845494f8741d9d57d890719bd54:action", "state_id": "9cac024ddb5e3b5ef7fb837b7fbe0d99fd027c727a044d5c5181ca6a46b888c0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.25, 0.75390625, -2.30859375, -1.5], "student_probs": [0.04128096625208855, 0.832395613193512, 0.03893166407942772, 0.08739180862903595], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "39fdaff340258283ee175ef3b8ce10c19c9e6ce590a7ce733e864f635ea5ce47:action", "state_id": "e4720186999de3efb41bb5b12fa2b0d4e26223c1666959fccf9ba9ab86560b14", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.46484375, 1.05078125, -1.63671875, -0.75244140625], "student_probs": [0.06151836737990379, 0.7612492442131042, 0.05180365964770317, 0.12542878091335297], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d41a22d92c9b3b3f303a9c6c644eebad9012084194dc240d9400fe1f57989a81:action", "state_id": "6b514351d9306c283368ed15947edc5146b1aa1f0a0cb1ec74ee5dee9602784a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8515625, 1.654296875, -1.640625, -1.041015625], "student_probs": [0.02645920217037201, 0.8813575506210327, 0.03267275542020798, 0.05951039120554924], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "07f90bb8de1f4c64a9126883c499b2636f99a209eb6a64b20c3228b7d18423d7:action", "state_id": "4f855dfdba60ab7a68358646fcbb7cbd64950e257a0fbd4eab7d59998f8883c4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.578125, 1.9765625, -1.2265625, -0.583984375], "student_probs": [0.024937298148870468, 0.8722290396690369, 0.03544304519891739, 0.067390576004982], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "042ed39472ce90d51e065544ecf5e4cfdbc035dab8639d3bd188950d51b11eef:action", "state_id": "6ebeec98b0833441984f411e42d848bb66ea8d177e43a66c5bc24c23ab4b34d3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.80078125, 1.732421875, -1.078125, -0.52001953125], "student_probs": [0.024454226717352867, 0.8371524810791016, 0.05037320405244827, 0.08802007138729095], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f271bb7f82e56bf9589b313f9a26e7c4b4cb6f051f100f5e0f558657062a3dab:action", "state_id": "8264a016ecef1500c220a4bb7ed23a0cf92fe1654dfbc3776cf4cac94e73f3fe", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9765625, 1.708984375, -1.65234375, -1.3828125], "student_probs": [0.02269599586725235, 0.9048194289207458, 0.03138742595911026, 0.041097138077020645], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7b2e0dde8a9f9b07c93d46ef777a9d5a0edfe87bf909ecce75df47c2e446b633:action", "state_id": "6cf5cc141458275013686e5a2d6b467e12e517efebe1500b62a0aaa7b8f687e8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5859375, 1.51171875, -1.6015625, -1.26171875], "student_probs": [0.039195042103528976, 0.8680128455162048, 0.038587380200624466, 0.054204776883125305], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "68ed4ab632bfbc7dea13599199514d021926898859f65ccc7b1443c11437c707:action", "state_id": "10385e504fd110f947fb9c16967fa24ffdceeb98d573b042d4ac9bcbc71854ce", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7890625, 1.765625, -1.56640625, -1.38671875], "student_probs": [0.02582537569105625, 0.9032912850379944, 0.03226599469780922, 0.0386173389852047], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5408d55f262dde7626c32871d77da468207f88272247e138e232decc39d7fea2:action", "state_id": "9606ad71397f68a3854ce5236096b7a341147d9f2cd804bb674c23a5318f711b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9765625, 1.8359375, -1.33203125, -1.328125], "student_probs": [0.019967615604400635, 0.9038032293319702, 0.038040176033973694, 0.03818906471133232], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "412ae18a359f0184ffb4c1fbbe455209fc10b54d8584d4f337d90e0c185b8b5a:action", "state_id": "6ba8d93ccef954839afbd9eb0288ff6567c4dccb9718d787ff8c0b5759b69821", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.4140625, 1.23046875, -1.77734375, -1.80859375], "student_probs": [0.023262731730937958, 0.8901445269584656, 0.04397280141711235, 0.04261990264058113], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0ef0c24762a1cfc82dd56b64da26d64ec3631060d2ed4e20bc396361853e77f9:action", "state_id": "9688da1024bb36e7b2bdc1df8ae6d8aaafbe97fc54b81d3182c23e5a1ad2ca32", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.03515625, 0.99609375, -2.39453125, -1.6328125], "student_probs": [0.04181203246116638, 0.8664758205413818, 0.02918950282037258, 0.06252259016036987], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "65c40f3275ca98460214b56f206403a9c25a093e761d5da1cf313cbb11eb966c:action", "state_id": "42652a0e9cc6885bdb26231ba8fa6916dd65329bd60748f4cebe504c52e12f2a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5859375, 1.21875, -2.0, -1.453125], "student_probs": [0.051746685057878494, 0.854954183101654, 0.034202467650175095, 0.0590965710580349], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "14b643186af1032bcf856d655deea573d01bbc1808925476bb9212135cda8d34:action", "state_id": "c64ef03d4acd88516b860d1d6fcb3b1d45bec4502f012f4ae00db28508d80946", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.59765625, -3.21875, -2.80078125, -1.71484375], "student_probs": [0.41887354850769043, 0.08280391246080399, 0.12576864659786224, 0.37255385518074036], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4acabdb4444cc01893b7a1af0d835245a93d4d106f651f3b0319d87bf88de3aa:action", "state_id": "c6b8151772a476e78d44e3fd3d8ad64fe80e958ee78f253a1ea6b7e10cf79cf4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.6171875, -0.134765625, -3.3125, -3.234375], "student_probs": [0.027501966804265976, 0.8948708772659302, 0.03729819878935814, 0.040328968316316605], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "febb6caba4ae5661a88c3bfc692021cac66419666fe26b90be31825cc88ee131:action", "state_id": "31d4e3337f5520c36a1d7a708e87fba456d7d3e9d8f9a3756fbd4d8018686f74", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.390625, 0.1328125, -2.56640625, -1.921875], "student_probs": [0.06286070495843887, 0.7839605808258057, 0.052727650851011276, 0.10045111179351807], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0c9c66c0ff7c442e0eb672cbbfd265bf6aaab6983b9dd44686de47f5ead9eb83:action", "state_id": "b92784ff900247ba00c63e6df756a21e4286dd0cbae2f0bd154672971ad8a2a7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.859375, -3.5390625, -3.6640625, -2.4921875], "student_probs": [0.29431918263435364, 0.1491537094116211, 0.13162767887115479, 0.4248993694782257], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3601a0751a2394579c1dedb2230247b1f8e163479a1f6401aaa9268ecd7cab88:action", "state_id": "53c9835f723cd8815ff8ba5827897684a715a8a7933916c6eae8abee138a4564", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.09765625, 1.0, -0.087890625, -1.068359375], "student_probs": [0.02993415854871273, 0.6629214286804199, 0.22335577011108398, 0.08378861099481583], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0eef11b275b3ee9de65476c412673477374ae911394033879c337c0d0edc5926:action", "state_id": "c80aa8be6a6fe1521f00a75a58bf6f0b01414cef4be66041ef7b05afa920236a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.578125, 0.57421875, 0.06640625, -0.26171875], "student_probs": [0.0540144108235836, 0.4647941589355469, 0.2797180414199829, 0.20147335529327393], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c0457f147eca7075b3d6d4e73cd9e85b8b4d35f0aa0d6e8d5191192812de59d1:action", "state_id": "923fc28d42f4eede5bc3a0ba6888cb9057dece589507ea7424b3c159e6582bc2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.255859375, 1.220703125, 0.0546875, 0.4140625], "student_probs": [0.045619942247867584, 0.5428903698921204, 0.1691679060459137, 0.24232183396816254], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fadcfb17bd63f3ab23cd67694226043744f4ef0e09bcb65bb955ac292326c186:action", "state_id": "c3488eeba0e5edcf862d460fd35964cd57c2efc2b1f7bc41d6218ab684d93190", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5712890625, 1.1953125, 0.30859375, 0.515625], "student_probs": [0.08178847283124924, 0.4785390794277191, 0.19716069102287292, 0.2425117939710617], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0bf43ae7f6291a62a675adc15d22e15b038f8a4512aa3c0c6c9759b2f4cd116c:action", "state_id": "07aa51348e78ecf15c172e2fc3ce1da03ae06bd9fd77de00c9f3fa18282e1cb8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5438232421875, -3.33984375, -0.03125, -0.1484375], "student_probs": [0.23721463978290558, 0.014482555910944939, 0.39604926109313965, 0.3522534966468811], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "84807b807494c81dd8d13b48a8b84ca7d520f1a44c256fe3f01760cc734a8980:action", "state_id": "59c4c3ac6988f054e4479db08bedca52b8d1c6586044a1eab28fc0e0886c06da", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.18359375, 0.8671875, -2.36328125, -2.6875], "student_probs": [0.016036996617913246, 0.9212021827697754, 0.03642337769269943, 0.026337454095482826], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7de2a43d25bf6c78e0630f190643114d0222ad647e02f79813fb75c51c0cbee7:action", "state_id": "da277c2d8c4f403da687d5f8d9ac425b1036ce58ab442ec001aa3b2c00c910d9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.01953125, 0.75390625, -1.6640625, -0.716796875], "student_probs": [0.0452084094285965, 0.7239487767219543, 0.06450559198856354, 0.1663372814655304], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1d4c03c77ea03ec39a3d4747108422e8aeeaa9531a86fc402913018aa1486794:action", "state_id": "2a48a6125307866d6951db6fb15bf26f02291d8022e2367bb3b1c642addfb2f6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0107421875, 0.96484375, -0.83251953125, 0.296875], "student_probs": [0.0763167291879654, 0.5503079891204834, 0.09120546281337738, 0.2821698784828186], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5f8b70bfb76a009076cf757eaf67824492e6e9136cc9311e46cf7fd69668f60e:action", "state_id": "d4d1a6bdcf416b010b063a97eb20f649d869ca577e8ef8e7f89303e3fbe72ffc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58984375, 1.26953125, -1.296875, -1.013671875], "student_probs": [0.046360187232494354, 0.809013843536377, 0.062141235917806625, 0.08248470723628998], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9bde9d86683e9d26c9f722c53fbb04d89a5a6b80480277d788e53f8d11e5bfc9:action", "state_id": "334b365f85ba7347dfedf77253fa4a9e2f0765e5e37f1438adde03dbd58b6ae5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.87109375, 1.73828125, -1.48828125, -1.123046875], "student_probs": [0.024083485826849937, 0.8897151350975037, 0.0353160984814167, 0.05088525637984276], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "06e08be7c67b35a34a988220676acd784c79cb61cc690fa58ce05859ac931307:action", "state_id": "68cd2f4dcbbc105acc4483e0d17a7d0166ce136e4536288a78cca9a51764bf4a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.75390625, 1.828125, -1.4296875, -1.095703125], "student_probs": [0.02483808621764183, 0.8928418159484863, 0.034349825233221054, 0.04797026142477989], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e10daba708a1d06901c90f8785aa705455a062411505a1cdcadbe182a69a031f:action", "state_id": "408d76046a376cb0d5cbde5705ce0d258705b29423e20fefc57d92444e001ec9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.43359375, 1.5234375, -2.1953125, -1.71484375], "student_probs": [0.01766074262559414, 0.923689067363739, 0.022412650287151337, 0.03623749315738678], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8b2b1c3f2ebb0f7b9503620e03c085fdb3253c51e5e4253cea374e883fc8b780:action", "state_id": "56bbd0e375af6d62cb266708c4a814f65eac3eb466801c077b35021d00a1c8a1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1015625, 1.67578125, -1.90234375, -1.3828125], "student_probs": [0.0208454392850399, 0.9109417200088501, 0.025440791621804237, 0.042772065848112106], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8059b429866674d1fcd5d594eba2df0d4c0d9ece2037629737a7112d54de31f8:action", "state_id": "222e5bd9022ff07ed83917ff9e2f7209f59bb098c42141c850eaf8ba5599631a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.54296875, 1.71484375, -1.5703125, -1.24609375], "student_probs": [0.03411655128002167, 0.8867784142494202, 0.03319631516933441, 0.045908838510513306], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81623d8c237f7e94d114e26daa1c03f19d01dc8c09e77cb0a579e6d32330f66c:action", "state_id": "9a7a2132c24f3abefe4517dc58482401d16681adc149f2f3322d79d49603c0ac", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.85546875, 0.66796875, -2.37890625, -1.7265625], "student_probs": [0.06578315794467926, 0.8204076886177063, 0.03897523134946823, 0.07483384013175964], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "62a14bb5bfe24a83b1079c7d41832b6638bcb1f239a13f8b12f4bd8f11cbc88f:action", "state_id": "6481afafa8f8ba9d203c4f48807642f850706484795d0fedfc435a894d630796", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.99609375, 0.5703125, -2.56640625, -1.68359375], "student_probs": [0.06269144266843796, 0.8161769509315491, 0.0354425273835659, 0.08568904548883438], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9d81741a37edb2338bfb9c2c21a1ba2ecb5e2db2dd8f36fb0461b4f83d24de6e:action", "state_id": "e46f19558cd18add0ee0144ff4d53af4d8ea01771d832bea47572cc3984d3880", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.28515625, 0.640625, -2.13671875, -1.044921875], "student_probs": [0.10461562871932983, 0.7177162170410156, 0.044644471257925034, 0.1330237090587616], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1c293d02ebd0e8858ae5668d2b44af44fcd84060c7c35f1a4bff700b10ac02b5:action", "state_id": "b04becc9792a7d28ea5eee739a6cf205787781667c6d0a6dd523a675d571ee10", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.62890625, -2.1953125, -1.7890625, -1.296875], "student_probs": [0.2622353434562683, 0.14883467555046082, 0.22342729568481445, 0.3655026853084564], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "742fc737dd7ba8ee92a3b2dc2b8d0e15f016856aa52aa177d7c4f34c68c26d54:action", "state_id": "e45c53e10772f7a629aa3c2cee05626925c04e53a380111d00a261a5524603d1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.232421875, -0.615234375, 0.96875, 0.140625], "student_probs": [0.1548442244529724, 0.10559459030628204, 0.5147037506103516, 0.2248574197292328], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ab36e57a4833fa1439d15303ab9efe1ef386bf88104db5ab25326e7b0319688e:action", "state_id": "13ce64258aac8d74d87fd104cd53167402a455c6f62f8dd8f7cbcf3d9d985a03", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.34765625, -0.2421875, 1.1796875, 0.9140625], "student_probs": [0.17811596393585205, 0.09874996542930603, 0.40930724143981934, 0.3138267695903778], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fb33f52c8caf8bf0ed8ebc834e0b720349e0bc0be6c4c661d26bf8cbe26b139d:action", "state_id": "352df0265001640647aa0112e76f35c9a43e4d09ae83a2230bf6d2d20994a98e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.015625, 1.16015625, 1.029296875, 1.072265625], "student_probs": [0.09948410838842392, 0.32239553332328796, 0.28285086154937744, 0.2952694892883301], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "101da8e0b08cc847eaea9203380b1ae55b5a46063192db599a063f6474bda521:action", "state_id": "fec0940c2f5a1d84b872f6c644a248c1398cf4f1692e6c1f4549f587c4f20f53", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.83203125, -2.88671875, -0.23828125, -0.3603515625], "student_probs": [0.22018755972385406, 0.028213264420628548, 0.39870813488960266, 0.3528910279273987], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e76faf7ae642b2c02842b9862962ebf7c3f8c2ed979a9db0477ed3bd76bf9ab4:action", "state_id": "9cc14247935968d0f259388525e54f5e215d7f57b231ad3cc808f502f6372c45", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.91796875, 1.09375, -0.451171875, -0.931640625], "student_probs": [0.03528706356883049, 0.717114269733429, 0.15298093855381012, 0.09461769461631775], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9bd302af00a0483d51ba6ca9b78eb01b1aed693c095c35644e5575e7a1a47872:action", "state_id": "054c840bd059b5a2c9a84daab8cacbe288adc1a3ff54673215859c1efcbc0348", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1015625, 0.755859375, 0.05078125, 0.42578125], "student_probs": [0.06588168442249298, 0.4221169352531433, 0.20855529606342316, 0.3034461736679077], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d4d89df75ba1a635e249ba42ed2fca6c0c633a0adae1b2f6b6dbdcf5266bc048:action", "state_id": "cb7c0b7f5b046fae8aeebde0e68cf78b9b16ac9b57922f765f78af8ce6eaf671", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.375, 1.58984375, -0.498046875, 0.14453125], "student_probs": [0.03654260188341141, 0.7086221575737, 0.08783251792192459, 0.16700269281864166], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8a82b32575e148a9003b12293adea4ad4895c54b68917670b8b9d4fe0b82fd63:action", "state_id": "14a3ed62126a1cdac8c73050d9fcbf1b8014d984897a7a9c23966db3508a6c57", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9453125, 1.3671875, -0.73193359375, -1.1640625], "student_probs": [0.02940942347049713, 0.8073967099189758, 0.0989578515291214, 0.06423608213663101], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7e199a2d4f5f63b92b4d7be4a73e4945a79e0d724c70a554e5ce4b466d588bde:action", "state_id": "12020904d4e854fb89fe70bb9c1e9f66e8771eb29ff83d80e66ad6ddf19a284d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.96484375, 1.64453125, -0.841796875, -1.37890625], "student_probs": [0.023356936872005463, 0.8628742694854736, 0.07180404663085938, 0.0419648140668869], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "292db2da6a7314450127c6a036fee1834e2a8b7d118467dfedf19435aca72fb1:action", "state_id": "3febbdd4f74458cb45ff40e7ea39d0f1f03e35067d86ee9515ea095ba1405c63", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9482421875, 1.728515625, -0.158203125, -0.466796875], "student_probs": [0.05165349319577217, 0.7509323954582214, 0.11381756514310837, 0.0835966244339943], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6c538111a95b281659bba19ffcb555612d4f78e12235897fab4d81741aea68c4:action", "state_id": "1343d4eb4e7c834a00c1535bb4c03db1e302372cded895771cc0136de4d2a341", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.790283203125, 1.7109375, -0.2890625, -0.279296875], "student_probs": [0.06055084988474846, 0.7385613322257996, 0.09995340555906296, 0.10093429684638977], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1e5533c9c3bb5df1782fa1a0d419f94a51e3feb43197df1524eba8417594ef3d:action", "state_id": "36079655dbb9d7b7400882595fb3cb2a47e2a8282b6f74afa7cc08d4709d86c4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.85498046875, 1.806640625, -0.91015625, -0.259765625], "student_probs": [0.055312108248472214, 0.7920408844947815, 0.052342891693115234, 0.10030411928892136], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3eb4102255296113e3d0223c55a332717fbc36a245a3bbc3646790f50f6df185:action", "state_id": "2c8139461aa8d5c3d1732415330e1b1689c91c8a459b19d385dc2256af0065e3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1171875, 1.44140625, -1.146484375, -0.63720703125], "student_probs": [0.06058839336037636, 0.7826589941978455, 0.058839090168476105, 0.09791344404220581], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "745ae08149a55d700e32fca9cc044571c6f8f284d094010439a7e0e47d041f51:action", "state_id": "a689cae24cb52289914e185ccbf6d9d5c33263da4650b048a9e0b2b5b922fc14", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, 1.546875, -0.9091796875, -0.5009765625], "student_probs": [0.05860195681452751, 0.7749506831169128, 0.06646960228681564, 0.0999777689576149], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7accfe9c938cf03ba830659ad71a33f5474baff5b8e0f0549b8a569e90f4b730:action", "state_id": "28c449955c9e87ae2c48fa0309aa66fac939b533c0d262f4245d6fc5d91a3634", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.955078125, 1.869140625, -0.7432861328125, -0.3359375], "student_probs": [0.04775321111083031, 0.8045355081558228, 0.05901775509119034, 0.0886935293674469], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7960f83ec13c7ec6af7e15f6ac3b1e74afd22d54e4d325e4eecd81ee1864a72c:action", "state_id": "787aeb69da51690cc7404a39cd12a2775e82ec277e990e841f0672ed1a4e338e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.98828125, 1.890625, -0.7249755859375, -0.373046875], "student_probs": [0.04556615650653839, 0.8108406066894531, 0.059291791170835495, 0.08430149406194687], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "46fc99df4537f52873d09fea7338271e34e889dc08205486491e303db308d7ab:action", "state_id": "d7018f45f3cd40c81bd5222d2bed6c3856cbce00860bd369ff5c0fcc4231eeb2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.40234375, 1.962890625, -0.4775390625, -0.5], "student_probs": [0.028631120920181274, 0.8285926580429077, 0.0721898153424263, 0.0705864354968071], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5c07fc639596219d709328b64828f193ecdbf66a73ba4db8f7d4dd6cce8d3b1e:action", "state_id": "9b9d7346149345c5b564c9eb37062b3adb8fb283df0272e59415819ea794aea7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.30078125, 1.78125, -0.32421875, 0.0234375], "student_probs": [0.03422641009092331, 0.7462261915206909, 0.0908818244934082, 0.1286655217409134], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cbb983b99b03bcbff55459f660ec437733fa38583a51592a75e9ab5ad310b8c5:action", "state_id": "3c34cafacef8b4d427d2d9bd17cef2e3afee0e601a62c77a77c89b03085422a1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.36328125, 1.931640625, -0.224609375, 0.11328125], "student_probs": [0.028188232332468033, 0.7603862285614014, 0.08802109956741333, 0.12340444326400757], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "acd0970a14377c2cf0fdc7a3b45bf12f1fdeb8ed407affa8c300e622c72f3fb4:action", "state_id": "fcd32359ad9a57fa2ffd65225d1f5935da2de15e3def9f6acd713a8a8f5e6549", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.806640625, 1.921875, 0.3671875, 0.794921875], "student_probs": [0.04080754145979881, 0.6247693300247192, 0.1319858878850937, 0.2024371474981308], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "38560ee87337bd31ac975c56068d4c8692eca3fc08a7ceab9a1eef4d2866c00e:action", "state_id": "96ce39c3805880219af46f837bf10e69b0ac4ba9aec81f52b23b009843364a7b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.687744140625, 1.986328125, 0.73046875, 0.892578125], "student_probs": [0.040841083973646164, 0.5921505093574524, 0.16866280138492584, 0.19834557175636292], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c18e7e2bacd238a7e92b7321cad26dbeaf1eed7657007195da832275c64744cd:action", "state_id": "6ad54d1a58ba28cfa7d536f1080b1fa73d8602a4dcacdd2d456437d48f90c32e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.1171875, 2.193359375, 1.0078125, 1.171875], "student_probs": [0.05621282756328583, 0.5666216611862183, 0.17314767837524414, 0.20401784777641296], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "964e350e2726fa0227d220005b7a8ed9d2d491683e889ee78e95eae9138d0baa:action", "state_id": "2fe151339bc2a7828c6c36cad5d6a45ed4f31f540c9d5b3454d3cf7fcd3035ff", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.201171875, 1.931640625, 0.78125, 1.05078125], "student_probs": [0.06407525390386581, 0.540703296661377, 0.1711396872997284, 0.2240818440914154], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e6295622838c0b7ad38febfad63dca1850b041ff0302898685a19baf981f870d:action", "state_id": "bd6d57835d861c74147e18cf565ae8583c5d6768ef0523e2a44ce23fd106fcf8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.27734375, 1.9775390625, 0.712890625, 0.982421875], "student_probs": [0.059699226170778275, 0.5691829323768616, 0.16070227324962616, 0.21041560173034668], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "52b21940bfe83c9c57541c070b8e09865820ac5e9d87f1fe9c57690a963386ab:action", "state_id": "f601a5994e28d45464c4a1631bd496d1c6317a16e92adf35e38c894216caf80e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.38671875, 0.92578125, 1.939453125, 1.716796875], "student_probs": [0.08912570029497147, 0.15279699862003326, 0.4210628569126129, 0.33701446652412415], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "df070cce4c09107e47bdf9989a9b13ebd0f07f4d76679c66beccd0a73f58d2dc:action", "state_id": "c0bfdba0a59292ae026f38165039e283acf9829730a9a3d5101985decf47c044", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4775390625, -3.37109375, -0.06640625, -0.087890625], "student_probs": [0.24750232696533203, 0.013706433586776257, 0.3733636140823364, 0.3654276728630066], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f27a00a9c255fb51ecf975dcd24f8ad2374c6be5e078ad3c4ccf36b900e9bf56:action", "state_id": "1291a014b34745c4dd6a9235ee0b4164e868b689a5303bfceca9c94075cddb21", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.2578125, 0.90625, -2.609375, -2.5859375], "student_probs": [0.014450281858444214, 0.9296205043792725, 0.02763688936829567, 0.028292279690504074], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "54b806c9c246f9a1e4fc93e2e55a87998641cf7be134c76693833bfe27962ad6:action", "state_id": "50e53a037ec9c1fe1fa990106baf10758d10a50dacb1eff427186b494ca08f57", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1015625, 0.6640625, -2.265625, -1.21484375], "student_probs": [0.04959134757518768, 0.7879552841186523, 0.042087629437446594, 0.12036576122045517], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0e7469c1437cdc8757e4fabe711f27452088032c26fa0cd5aa93b7c2090c8518:action", "state_id": "0f3aeb77a61e48808c5d119b46ef0231da444035a92b850408d8ae7c807cdc31", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.84375, 0.375, -2.44140625, -1.5078125], "student_probs": [0.08233719319105148, 0.7571586966514587, 0.04529364034533501, 0.11521044373512268], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3f4b4d2dfa69ba652089725b9955c69918727592f7be2e5e59e5f7a08efd83ce:action", "state_id": "3895480fe661763048577557d558386e8975a25ec959a69285391c2b3186b420", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.5859375, -3.1328125, -3.3828125, -2.28515625], "student_probs": [0.2958225607872009, 0.1712089627981186, 0.13333767652511597, 0.3996307849884033], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0f3905f323ec1f7af84a324d17d73f4ef165f3de721a78ec1c9ccc4b10d65fc:action", "state_id": "427fd63577851fb757bd18a4f455a2eed1f5c0949aa0cc68ef5e1d6f5382ecc2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.45703125, 0.015625, -3.578125, -2.93359375], "student_probs": [0.027936091646552086, 0.9001628756523132, 0.024750005453824997, 0.04715108498930931], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a71ab283c5a8a8a51b33a2627412c7b826f493ee77704ccc4f8393ef1bdeb2c0:action", "state_id": "6d92fd70bc600813d4cdf92dae7b963c34aa15493436beb686976ab79906eece", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.73046875, 0.29296875, -2.62890625, -1.34765625], "student_probs": [0.09580479562282562, 0.724694550037384, 0.03901223465800285, 0.14048844575881958], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "25448438f209eefcc8de862dd0f358130038f96bf4728f06b267a162401c5aba:action", "state_id": "161f50ab29f0762447ba49789d4c7ac2f230aff31f4bc7d6c59a8a6059446d48", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.09765625, 0.1015625, -2.75390625, -1.484375], "student_probs": [0.08075430989265442, 0.7282395958900452, 0.041894786059856415, 0.14911124110221863], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5ea5716fcbb2cad265d07305bad65384ecd23e40aa60cc35ca9082035b0ecccb:action", "state_id": "a2b5b407abdaeb919c316173e234ca6c616ecd6d819658e0598e0116d8faf04c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.40234375, 0.29296875, -2.75390625, -1.88671875], "student_probs": [0.054980043321847916, 0.8142624497413635, 0.038683291524648666, 0.09207423776388168], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "54e22628d59e45deec9d15941dff20a543c11fe0438cdcb9947cbd5e54397cc5:action", "state_id": "bf1676f385f757dbd8d127429f7120bfe4101e96f433925fab4c8c59d3ea2d7d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.60546875, 0.90625, -2.8984375, -2.28515625], "student_probs": [0.02730046771466732, 0.9147241711616516, 0.020367387682199478, 0.037607982754707336], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d8befd2fe7c1051ff01f3d0ae40521d35f50c1efc12f466fca9c7ede2bd7b526:action", "state_id": "ad6e9e0c5f9810d0dadcaa76bc121e3d9dd68a66acc714f3bf838edb5ef4074f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.47265625, 1.79296875, -2.1640625, -1.97265625], "student_probs": [0.013294399715960026, 0.9466863870620728, 0.018100446090102196, 0.021918760612607002], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "beacb3fe1a793c3d712b6904fcf43f197cf2bb7496144dd07dfa07ca43791d70:action", "state_id": "89b04dda309bc71775f0cc00a96ab5c62d74c156a1fc07300fd721e0a34658c0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.95703125, 1.548828125, -1.56640625, -1.4375], "student_probs": [0.0266885869204998, 0.8889983296394348, 0.03944317251443863, 0.04486989974975586], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0f37c6e69690a58717b4268237992b2db79b38e0f6fc4e32c74135fbf9e1b58e:action", "state_id": "05d86d0c99e87ccad8925aff724ee77aa189f55416b365def1dbad0550d93247", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.34765625, 1.78515625, -2.08203125, -1.42578125], "student_probs": [0.014887312427163124, 0.9282692074775696, 0.01941671594977379, 0.037426698952913284], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2e0dd15f37328d22f85dc1054e9c17c84e0d72bf921e764991814072b391a88a:action", "state_id": "22b29409058597d81a4f97cb0228842be86d1fe6c960f4a96850b890b9adc082", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.83203125, 1.57421875, -1.9296875, -1.37890625], "student_probs": [0.029733460396528244, 0.8965222239494324, 0.026967080309987068, 0.046777304261922836], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7e764e7050364864326e5ef09ad3da0771265d5af9c79b8a1d4d752a8de56d2c:action", "state_id": "ebd9b279df007f6cd78a32680be532fe3e91e21fb32d88de921a382f94888529", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8046875, 1.68359375, -2.01171875, -1.203125], "student_probs": [0.027496999129652977, 0.8999670147895813, 0.02235490083694458, 0.05018114298582077], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "678db23eb99e5888e65fbef4ffd8518e0ade8855e46af8f069dbf28b4100bebb:action", "state_id": "bd4a053ee91751b08ab1e383e5e4b9c461d6b65bfef8c5b0eed2d50f8a7d30ea", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9453125, 1.34375, -2.234375, -1.4375], "student_probs": [0.033081550151109695, 0.8871714472770691, 0.024776935577392578, 0.05497003719210625], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f48b117d2c6dfb28b42a3ff3882f3f91c0942cb3332f6d405841454bb15b4e78:action", "state_id": "2c05a1aff879809aa2700bf765fd60e3934bb6f5f03d76aead02a72cf83c9f5a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.69140625, 0.69140625, -2.265625, -1.4453125], "student_probs": [0.07311277836561203, 0.7922014594078064, 0.041173070669174194, 0.09351266920566559], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "135fd77392ee7ed9ef1776858a27cecc08815ab97d440556dbd0e69d91c01310:action", "state_id": "11dd8cd295961548875ee10b53c49ae8d4cff837f10e4ad59ba599e804678183", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.671875, -3.05078125, -2.87890625, -1.53515625], "student_probs": [0.37072139978408813, 0.09336761385202408, 0.11087679117918015, 0.42503416538238525], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9b190c42d70f6b6fe65130dd809100b3e3f56ead6acd5c6488c72cd9938002a3:action", "state_id": "47432a7636d9139e7d28934a906c509d259ff6ca42586cd512bce57c2fa616c4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5703125, 1.216796875, 0.16796875, -0.435546875], "student_probs": [0.03841421380639076, 0.6236173510551453, 0.21848315000534058, 0.1194852888584137], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0ec2b144946db82c6f3acca63a69f562fbd851324c5ef0033e9929819b6677c0:action", "state_id": "07ddfda1753a69bc56520a4c2e6e2b05b8ba613be962cb4f63e616984fac68fa", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.158203125, 0.69921875, -0.09765625, 0.2734375], "student_probs": [0.069057397544384, 0.44246435165405273, 0.1994343101978302, 0.2890439033508301], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2ed92050ade1614cb8c9341c52650a540784636536776ec71d760ddd82cb7714:action", "state_id": "43b970d1fca86b87d9fe67638d327dde6c82071e2243570b3b10fd026d8cade5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.42578125, 1.38671875, -0.443359375, 0.02734375], "student_probs": [0.04065227136015892, 0.6769211292266846, 0.10857884585857391, 0.1738477200269699], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0e7ce3db5a2e880c1007af1c20b303978ff690bd4f3dc786bb36798093c16146:action", "state_id": "6000d6f8fb1a28cb2590db46aef8f79761255142cce845fb33c44f66dcaab028", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.75, 1.365234375, -0.65777587890625, -0.8291015625], "student_probs": [0.03444575145840645, 0.7763628959655762, 0.10267923027276993, 0.08651208877563477], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "80d88ad87f21a9d453690c2b3cbaf7bee1e49d3335fe4bfc56f7ef780543e1f2:action", "state_id": "95a1f6d58cefbc2f4ce51a52982c01c8ce34915bd0637239d85e555a2c1f1dce", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.96484375, 1.787109375, -0.775390625, -1.2578125], "student_probs": [0.020442601293325424, 0.8709408640861511, 0.0671597495675087, 0.041456811130046844], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e57523da6902bea81812160de75b781c4917d723f1f9123833bca7ed26087ba9:action", "state_id": "3b4d4e4c00202985f114ca8cf41820ccbb4f943bc82b8db88b4d8e5a08af2e58", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5546875, 1.703125, -0.388671875, -0.7060546875], "student_probs": [0.03073306381702423, 0.7988327145576477, 0.09862794727087021, 0.07180627435445786], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d3d89eb49d2c9c95824dbbbb746e48d0419acdcbf05a04cd830b15d863ea0d95:action", "state_id": "b018d0c75e63cb7dfb8eea4daa0f637bdb1dd8be265c8e9f176fd3e76c0e1477", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5390625, 1.953125, -0.709228515625, -0.2109375], "student_probs": [0.025047186762094498, 0.8229940533638, 0.057431645691394806, 0.09452708810567856], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "806ccc9ca19ade114e065d6e243641003461c93963578176c0f9f48a60630022:action", "state_id": "6b95b62ad5eeeb7c144d8995f182a05702102f0532724e40bdfcf95b22f2ab47", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.26171875, 2.05859375, -0.6834716796875, -0.1328125], "student_probs": [0.02981143444776535, 0.8248524069786072, 0.05315111577510834, 0.09218507260084152], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "16515e67aa40a511ae96c3d16e8c006a6b84f532f5e00002cbb172f4466cc64f:action", "state_id": "de23e48c20316a38c34e04514617a0c0afb35d6a2e0e1a9d46859297dcafe37e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0283203125, 1.712890625, -0.76165771484375, -0.41796875], "student_probs": [0.050884295254945755, 0.7889991998672485, 0.06643452495336533, 0.09368198364973068], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "230eff0c1272538ad588b063decc390e1bd82a0837c4b1733c50a7753b9f6c27:action", "state_id": "8282114d4321fa133c34c1d8f8056fa4d6eaa2ea7cdd849189027e29ce71a24f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.185546875, 1.783203125, -0.7187342643737793, -0.396484375], "student_probs": [0.04121365398168564, 0.8023296594619751, 0.06573175638914108, 0.09072491526603699], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "83ec1491383b16ab458376e199dc99bc9c0764ebb7f149b0e2f0a7185b7a3e3d:action", "state_id": "c8fedd4873813529f1954a1ac4a1488fa7bc811710a20e639523d5dbffbca4e7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.09375, 1.951171875, -0.5751953125, -0.181640625], "student_probs": [0.03820066154003143, 0.8025344014167786, 0.0641617700457573, 0.09510315954685211], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9c9802069a62326e172480d40f0c9a67f900c465510417b6cdd0a798aba9c221:action", "state_id": "af32cf9d9b4797224d7b67f0196e1ef578df0fda642bf46e18fe285b53ce656d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.15625, 1.685546875, -0.3671875, -0.150390625], "student_probs": [0.04332354664802551, 0.7428492903709412, 0.09536949545145035, 0.11845767498016357], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bc4c22c0bf4044aff09f890183b3b9a82080d8c1967828bed8e4584e03352e71:action", "state_id": "d97539ae87a052b6b1398edc75f20f916e1fd96275f94cb7a3e18324601e080f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.150390625, 1.86328125, -0.10546875, 0.078125], "student_probs": [0.036203864961862564, 0.7371841669082642, 0.10293397307395935, 0.12367801368236542], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "846c59e76f48a6b0bbcdc76fd1815a78d0515990643f2b02af99ae765695ef00:action", "state_id": "6be336c6dfe75ef8502950102ca66d2a067dd52ab273a9ef3f79565fdd7e2826", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.06640625, 1.923828125, 0.2578125, 0.3671875], "student_probs": [0.03467005863785744, 0.689599335193634, 0.1303333193063736, 0.14539732038974762], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cafefe722fa273d80e16c9d39f5ef085aec827d7632cead28fb1bff04a793711:action", "state_id": "35142f69623bc9bd92c39dfe435d79d851ced091d315a382cd43511825afe02e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8740234375, 1.91796875, 0.51953125, 0.55859375], "student_probs": [0.039166100323200226, 0.6389356851577759, 0.15780597925186157, 0.1640922576189041], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f66e6caaf670c261c36f1d3afe9b4f0c0eaa5e5bac9e2f285bfd29e1dc7ba029:action", "state_id": "08c9e29e7734f0e711f21c52de78a7972af6d374b4456862fea28c266e8a4dc6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.357421875, 2.1025390625, 0.935546875, 1.134765625], "student_probs": [0.048089053481817245, 0.5628513097763062, 0.1752166748046875, 0.2138429880142212], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "78c4e15ab751bd9be572f5f24f3e8c1acaad85007c1d0bb71235afe3d1ec7efb:action", "state_id": "f37437c7c3b5e4e9a0fd7f01fbc2bef7a29b848465849732258d5124a904abb6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.37890625, 2.0185546875, 0.888671875, 1.052734375], "student_probs": [0.05067646875977516, 0.5571991205215454, 0.1800149381160736, 0.21210943162441254], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e1f03fe7eec6cbefc336a3ca8e046e24722f64f4639584667b918a6aed548e3f:action", "state_id": "4e6f3345111b4cb5c70d476961cce1cd33bc2ead13bfa8514fffa1bbe3ac1da6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.02734375, 2.1298828125, 0.994140625, 1.1796875], "student_probs": [0.06341966241598129, 0.5483976006507874, 0.1761363446712494, 0.21204642951488495], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "33c8a798b27ffd3753b6ede282759de629c3127a8b1a66bf264b89add2b1bec3:action", "state_id": "f19c9dc7ebd6fa66fdbc41fc9d5c1cf061670db0d9aa533512b8158a5b83d066", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.588043212890625, -3.0, -0.119140625, -0.080078125], "student_probs": [0.2298964262008667, 0.020607849583029747, 0.3674294948577881, 0.3820662200450897], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "db9229317126e673ae53f23b917f420268896c5c040fc7167cb84c4d058da007:action", "state_id": "49402c9ebc58c39afc8989def66438b7f654141c05f399d16c01b9db22a93159", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.236328125, -0.69049072265625, 0.935546875, 0.09765625], "student_probs": [0.1597561240196228, 0.1014418676495552, 0.5156990885734558, 0.22310283780097961], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fc851f121c2f3a52f8c63cf06cbe7c9bf151ecb6830ed85b13c86c04b1ea1e29:action", "state_id": "c0a32fad692ad95d5d900ad17029dea5a936e431c9ecc9236244ba0441210f44", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7578125, 0.0703125, -2.1875, -1.36328125], "student_probs": [0.10687632113695145, 0.6650068163871765, 0.06954574584960938, 0.15857118368148804], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ff25789c3bbb3af37230acd5b0d5f1e12a299e89b9ea41b0bcdc13afc3ed52f1:action", "state_id": "f64af29f4df0592455ce677b97143e3db7e74004a678968dbac22d6374295305", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.45703125, 0.16015625, -2.34375, -0.91796875], "student_probs": [0.1224694475531578, 0.617111086845398, 0.05045807734131813, 0.20996145904064178], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b6cd9747e098e2d018f5b932c06d59b7d894965b88bcc8da3462e84fa24837ec:action", "state_id": "7f72808818bb66c7e4e666c2e423ce90f02f525da9959343d78a3bc92f2e5f84", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6796875, -2.10546875, -2.15234375, -1.421875], "student_probs": [0.2800571620464325, 0.18294991552829742, 0.17457203567028046, 0.3624208867549896], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f86501710b082327c51cbe2c8323a83196e770a81000a0fb055445f8ee0833a4:action", "state_id": "0126c485d974cedb4d1ec5fe36a6e3b6c8bc3a51c2857aacae71d41a4c1fec30", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9140625, -1.3515625, 0.919921875, -0.953125], "student_probs": [0.044678136706352234, 0.07841256260871887, 0.7601141333580017, 0.11679517477750778], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b1843ece326f2de53aebaf364813105465e97a616806005d8e43157ff6cb85cc:action", "state_id": "02f73a7f9e3d71418f55a0888cd01d89cc567bc51fabf0a38fd4d9d923b85b71", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6376953125, 0.728515625, 0.875, 0.58203125], "student_probs": [0.0778472051024437, 0.30519744753837585, 0.3533444106578827, 0.26361098885536194], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2156ae4c81439806b271d142025dfa48bfc0d40c335f653eea18c46556e00a9c:action", "state_id": "ea07ea0b9cac8044e65b29439c572f2aa3019a8a4f950d3d0a1a427fad1151fe", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9375, 0.994140625, 0.28125, 0.34375], "student_probs": [0.06718210130929947, 0.46361178159713745, 0.2272741198539734, 0.24193204939365387], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8f20a6a7ba01901482876f8b9989e72c6c2f8e463b6fc11df20a0a55486225ce:action", "state_id": "0bd70230b54712bedb27d18de8435f27d3bb35c6567f2971689801f0410b1d81", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.107421875, 1.38671875, 0.37109375, 0.16796875], "student_probs": [0.047443170100450516, 0.5745994448661804, 0.208106130361557, 0.16985127329826355], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a5c7fbaf006eade7c3dfe996901333d0d16415d9d5610bf40b928e8d05c33513:action", "state_id": "438485e7d2af002b597d2c3e5c6afdcf5cc3ede90030bb593b0f129612610fb3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.48046875, 1.80859375, -0.458984375, -0.34375], "student_probs": [0.02966342493891716, 0.7955051064491272, 0.08238465338945389, 0.09244681149721146], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8805c3989c708100bb377efb16a7f8a6e7093a3f99dfec1eb9bd035008e2a565:action", "state_id": "5e13b9c9932875107cafbda7256ebbaebf028af5821a053c75e77cf8e9a49eb4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.42578125, 2.056640625, -0.478515625, -0.18359375], "student_probs": [0.02526511810719967, 0.8220873475074768, 0.06514988094568253, 0.08749768137931824], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0ef7742968b2e4750a9b592f48a48ff907c157740bc355eea3c57af32cbece56:action", "state_id": "cfe719db46b56635eda628302fccb09aa096a13dbff9a9869828c5c60e25274f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.39453125, 1.8671875, -0.55517578125, -0.15234375], "student_probs": [0.030420655384659767, 0.7938071489334106, 0.07041999697685242, 0.1053522378206253], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aef02ffdae233d3ec09f4a33b2462e8841c99a993b6b044b7b272d4d192968d2:action", "state_id": "afa20aa1fd77b56c95838dad01358acd29934f668621a5e9a8682d196c316135", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.52734375, 1.96875, -0.8779296875, -0.5185546875], "student_probs": [0.025877881795167923, 0.8536167740821838, 0.04954110085964203, 0.07096435129642487], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d20d1c6e3ea83733da407c16a06542032393c861f793f7898dbaf1c70916fe1b:action", "state_id": "29d1a548885ba0386a4380d70acc19cce73961ab20524560148dae63efe02dba", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3125, 1.681640625, -0.853515625, -0.51953125], "student_probs": [0.040386732667684555, 0.8064500689506531, 0.06391063332557678, 0.08925255388021469], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8d774a94f10ce48b2721b04ff2c261ae3400666199b85867d61beae2be1b526d:action", "state_id": "f467d8361625608576acc8140f1c1d722be0c7a972adb8fd68c57a49dcbb117a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.171875, 1.732421875, -0.73834228515625, -0.392578125], "student_probs": [0.04352549463510513, 0.7944449782371521, 0.06714668869972229, 0.09488292038440704], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b768510c046b5f1910a6ed6af11e1c0171a19c18ebe100649ba72abdb92d1205:action", "state_id": "eda5cf70f4a56525aa8bd90c2250ab772ae7be3ce51f1fe731709664956e186c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9755859375, 1.98046875, -0.541015625, -0.125], "student_probs": [0.04148120433092117, 0.7973511219024658, 0.06405939906835556, 0.09710825979709625], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4cde332a5f291f0ddae59c2063b62118d55b53e7d7675f6c9cd52668ed6f3750:action", "state_id": "aae3c2f8a3314794ab8030f5714a326a27ce63b24687133614689f1f322208e0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.23046875, 1.673828125, -0.4609375, -0.2109375], "student_probs": [0.04135126993060112, 0.7547601461410522, 0.08926722407341003, 0.11462138593196869], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "546df12d76dbf2993b4915d59d499a3df198ef16c4322b1d8dca9e4857df7e2c:action", "state_id": "c7c2bce1c5d88d07a00645fba56ce23d44f662f848fb2eeb30e0ba476fee3713", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.28125, 1.84765625, -0.234375, -0.158203125], "student_probs": [0.033588703721761703, 0.7674673795700073, 0.09568530321121216, 0.10325860977172852], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "26d772a82c13a9245f1e28c7677bca4f1d569d21dc9c0a71a3cf1517a7858aea:action", "state_id": "5e8c8ca5011589897ebcefb4af090133ab98640c346cb571375644ace3cf763d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.234375, 1.869140625, 0.03515625, 0.15234375], "student_probs": [0.0324285626411438, 0.7223828434944153, 0.11541921645402908, 0.12976931035518646], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6b0c46bc7e141daf7702a3d9d62fb27809d694a4352727c36ba76c9fa73ce353:action", "state_id": "73ea0bfbe93389527cb10f2e3b8f9b46f70e5fdf6ced039d114ae31a911f98fc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2265625, 1.81640625, 0.1015625, 0.25390625], "student_probs": [0.033182501792907715, 0.6957507133483887, 0.12522944808006287, 0.14583727717399597], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c6592c67c0e942a0145a0dd4a1e0c073fe20bba10f85b68f4a54c6a96c682372:action", "state_id": "de4a01788784889bccae9b6f2c25bcb2c4a9b266b5a1e9ce9a049f2341e8371a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8583984375, 1.9453125, 0.5625, 0.890625], "student_probs": [0.03650219738483429, 0.6024974584579468, 0.15114973485469818, 0.20985062420368195], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d1d3584230b0f8fda4ab3e6cc087f8909a3dc6d98f87db851601e05013fc2437:action", "state_id": "1270aae78e0dcb4023302dce2a5c09154be827e7057a85b4ee4465de167fbd6a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.529296875, 2.01953125, 0.87109375, 1.080078125], "student_probs": [0.043766409158706665, 0.5598644614219666, 0.17755088210105896, 0.2188182771205902], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8cfabb046071ea2837dbeec5dc5926b78e97c5b8fbd5c485aae8238dfa0c0b98:action", "state_id": "5ca00458b50f25398ebd03b79912aaea1d61bfb43d89b536bb8b66c11f77d0a6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.09765625, 2.23046875, 1.0625, 1.27734375], "student_probs": [0.05433543398976326, 0.5574102997779846, 0.17335350811481476, 0.21490082144737244], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b8c488ce2f8e2b94cfc541ba16f9783c3cd147b87904bdae25609f2711ab659f:action", "state_id": "31f2cd3457cf625e5a000f42df004293284c3df2ebe59cd165240db62ea2a84a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.37890625, 1.931640625, 0.724609375, 0.970703125], "student_probs": [0.05570844188332558, 0.561537504196167, 0.16794680058956146, 0.21480724215507507], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5c044feef081dd39becd38a3a5f836eb1fee3f2de2fc96dc453df02c06d554ee:action", "state_id": "e421a580c26d0be92511851d6341e3c85bddf62761b835e92f363402699a44ce", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.26953125, 0.92578125, 1.6728515625, 1.48046875], "student_probs": [0.09659159928560257, 0.18618518114089966, 0.3930009603500366, 0.32422223687171936], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "62a124198bea97a2b8aec6cc8829b891d8158f636c6370b57b60ffbed77b3e2c:action", "state_id": "5732df01f9af591939b9e8312831d39640447bea3d9677b3a6343241d1ad69f2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.24609375, 0.7578125, 1.7216796875, 1.7529296875], "student_probs": [0.08654873073101044, 0.14437676966190338, 0.37852931022644043, 0.390545129776001], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8e9ec82394684065965575f5b5d60c87f2b12bfd4f29e5c2671d7851ed63ed0f:action", "state_id": "6cdb0003374830548d2d0a18c1844b4005425003781d674702ee7490d6538da4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.46533203125, -3.19921875, 0.08984375, -0.03125], "student_probs": [0.22984497249126434, 0.014932175166904926, 0.40044674277305603, 0.35477617383003235], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "767c2d93e701405927113e39ebb7d8de11b1374d3eaa7cc1e9963fc8ae2e4119:action", "state_id": "75ff2755af78578b0c54fe08340f4bfa187a63db2b7ae1da23dccace296e501e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37109375, 1.18359375, 0.26953125, -0.5126953125], "student_probs": [0.0467616431415081, 0.6016950607299805, 0.24121491611003876, 0.11032843589782715], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6f63e63eb56e76fdce4003b31454a0b57cd5ce82540bb759055c04e4adb5548a:action", "state_id": "78695066d99e123b69fa7117327652ee1fadb0963ff393348aa9c0f40ad59327", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5869140625, 0.802734375, 0.453125, 0.50390625], "student_probs": [0.0924258679151535, 0.3709455728530884, 0.26150307059288025, 0.2751254737377167], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a768f932bc9e073fab2a2cb30831c16e7e7898efc178e92dc3c0ab9bdaa7140b:action", "state_id": "3db4caae4694327c01c091f922dd9aff59903dbba33079bf55e4ba0d6e10b829", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5029296875, -2.765625, 0.0078125, -0.03125], "student_probs": [0.22866128385066986, 0.0237966887652874, 0.3810703158378601, 0.3664717376232147], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ecc10d646df6421dd664ebe6b422e5f51d9141f757fd29923fa1d2b43dd12678:action", "state_id": "005fe06dc391eb501a33c8491c26742bd7a8cb2e5851f67c4f2f164f69c8d767", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.92578125, 1.09375, -0.451171875, -0.9296875], "student_probs": [0.035015594214200974, 0.7171785235404968, 0.15299464762210846, 0.09481117129325867], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e0f3f38b3f6cad97006704a56a5d67fff7ada01b992e771145001b665ce623c7:action", "state_id": "f806ac69105b0b798cf2a842db2fca4aca75f8d2fc2491b9a7baffb57f1da1d1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.916015625, 0.68359375, 0.4140625, 0.484375], "student_probs": [0.0725204199552536, 0.35905569791793823, 0.27422428131103516, 0.29419970512390137], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c50b790a9a6d00bbcdf6bacb7be769100cd143f19ea1e76a25b88659f42f23dc:action", "state_id": "9079f4b2c4b0697bdc12c8ffccf88617aa513de407bb5ec83f037c9c5676f819", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.705078125, 1.515625, 0.48828125, 0.658203125], "student_probs": [0.05740215629339218, 0.5288923978805542, 0.18932048976421356, 0.22438494861125946], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6523cc1b60651db42698968ff49b95d40abaf6507939ce538e8f6715702adc7b:action", "state_id": "86132836e6e9b7b31e16e0761b2399f572d76e7dc8d0cb50b57bb2f7568b548e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.779052734375, 1.98046875, 0.41796875, 0.43359375], "student_probs": [0.042616844177246094, 0.6730173230171204, 0.1410720944404602, 0.14329366385936737], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7cbf157f3691ddb8c4675ecdcb143bf5f71b01910fc7559f16df8201d06ba44b:action", "state_id": "848384d01275d2a54e6bdb533315adebf134857bdbcf7ebe20d8ea8d6f8105a6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.046875, -3.7265625, -0.2890625, -0.4365234375], "student_probs": [0.19828417897224426, 0.013599238358438015, 0.42305988073349, 0.36505675315856934], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7236695c48a5a9492be678774affb710ce5c4d9829e2b6ccca21c1c7001e4911:action", "state_id": "0212a6ecc21ffdb4a62dfafb5e393b5d39d4578e737ef2045c9a279b54fedb89", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55078125, -0.44921875, 0.421875, -0.7529296875], "student_probs": [0.07451910525560379, 0.22421781718730927, 0.5357736349105835, 0.16548939049243927], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "74cb225ec1cc32df652bf25de2e912386096c3bdbbcfaf1a75e04fbd06c30564:action", "state_id": "af0cec8b48e4f2d7cc8378507cdd3fc2e9901e1b0e7be702eeca3e94437d9eba", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.02734375, 0.21484375, 1.083984375, 1.224609375], "student_probs": [0.2688106894493103, 0.1192840188741684, 0.28447574377059937, 0.32742956280708313], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81de4c916e7de610fae18b9233e37228e5aeaab50347d058d65f769a31b84b4c:action", "state_id": "d5c9adda5fda9f8fe468b3c3a61443929abd637ffb348c52d576c1bf2eefaf9a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.68505859375, 0.986328125, -0.09765625, 0.484375], "student_probs": [0.08819107711315155, 0.46913591027259827, 0.1586829423904419, 0.2839900851249695], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "acb2b3727c98d0931c3a386afcd203251bd1b3c16d0acd3cbbe5b42b3c151957:action", "state_id": "32ebb4f2a71cab389a4f73b509c19e2823094241a73aee4b7daf734e525203ab", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.55859375, -2.546875, -2.87890625, -1.91796875], "student_probs": [0.21573221683502197, 0.21827518939971924, 0.15660478174686432, 0.4093877971172333], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9e17995b6ce448f7cd6350e597ed57f91cbdabdda6f2cfa0fa1f8e66c98ccb74:action", "state_id": "4aa88bd4a7078f9718fe9de6f3272d6bded27f1681c5ff1709421ca61a8628c6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.86474609375, -0.986328125, 0.642578125, 0.0234375], "student_probs": [0.11323920637369156, 0.10027539730072021, 0.5112336277961731, 0.27525171637535095], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9426a3620a7d2365250a770b374744b7a60a7f654239a897fd5a23e211a48a09:action", "state_id": "7ef1814bf348ce87c05f0432449daa481bb5ea0e016c58570b08efd7ba00a21b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.05078125, -0.359375, 0.806640625, 0.78515625], "student_probs": [0.170150026679039, 0.11290246248245239, 0.36232438683509827, 0.35462307929992676], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cd60ed56c3066cfcfcbd57e4f5ba78cc7115e21d426ddacfd3883766dacb0484:action", "state_id": "84b8e0236a24674c28d69a4cfe04fb3d2c249a3509c279713f0c03dd86dcb952", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.251953125, -2.9140625, 0.1328125, 0.04296875], "student_probs": [0.2575930655002594, 0.017980210483074188, 0.37847375869750977, 0.3459530174732208], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bf13da5e78e4b759b20f1596af7c0efc6f5525e94a1a55e4391a5377d9e6866b:action", "state_id": "14e38383144b586f705ec5a0e977c3bf35756fb7a7a365b9f94ab497bc72924f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.224609375, 0.921875, -0.8095703125, -1.06640625], "student_probs": [0.08169558644294739, 0.6988837122917175, 0.12372223287820816, 0.09569854289293289], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8f485cc846fca93ff01e4f2f2706df8e4f2ccd76450148963723936b0a9eca0c:action", "state_id": "2e5b849cb2c45ebfe696b3b27a2cee11edeff3dcb2a9242aea92ae41c0b33eb5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.287109375, 0.2578125, -0.69287109375, -0.738128662109375], "student_probs": [0.10833363234996796, 0.5078253149986267, 0.1962626725435257, 0.18757830560207367], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f38588ab81218e7f90d602d378d9c4cb433d52038d3f0e4ee14bde89a2f32b76:action", "state_id": "c06cb2584fb1214806b0c4bcf4a7ee5a02489d8456d4d137a22a6f2f08eeeed1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.36328125, 0.87890625, -0.9990234375, -0.859375], "student_probs": [0.07402734458446503, 0.6968861818313599, 0.10655831545591354, 0.12252815067768097], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fdb43e20f99411063a9162af4c466b54014da7bf38b5eb6fcfafc94daabea749:action", "state_id": "4289bfaffa835e4d5abaacd428a5d65ec245a7707fd18cdab3dd7f8b7ac0f3d6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5234375, 1.2890625, -2.0546875, -0.787109375], "student_probs": [0.049194153398275375, 0.8191562294960022, 0.028919752687215805, 0.10272987186908722], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "79c319272b672f54466b9c1bf71bd81d9e5a303babe4f8f7d667a0e705cda6fd:action", "state_id": "65a0d8fcde7c5591c11da0be5d834705670952bdbdf445adb7bfc5e0f4002461", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.28125, 1.37890625, -2.4375, -1.7890625], "student_probs": [0.023607928305864334, 0.917579174041748, 0.020192932337522507, 0.0386199951171875], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "297c062798c345d14c168ae5b2f1354a8f43e268b3775b7b14bae6d980bd8c05:action", "state_id": "78c257fbb1d71e0dd9ac082a9c0f12404c35d3545e2f9723e4d10270036f9a03", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7109375, 1.55859375, -1.42578125, -0.623046875], "student_probs": [0.03164859116077423, 0.8323265314102173, 0.04209166765213013, 0.09393323212862015], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "328bf7cddcc511ed949893417f6aa1b177979a629a7e246bf9387f9ab94b4734:action", "state_id": "a0da2b4e5bdf02bcd350882665574c0a37714433065dbb55eb144b9629ea96be", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.40625, 1.57421875, -2.10546875, -1.60546875], "student_probs": [0.01720568537712097, 0.9212290048599243, 0.02324339933693409, 0.03832188621163368], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fbdb464b18544181c017fd332736bec16b5f8ef43dbbe378829d55df3800005f:action", "state_id": "949fd930134d11415925295406f2a19e4ab3e11f29b6e33a89b64d092b463dc8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8828125, 1.751953125, -1.8515625, -1.150390625], "student_probs": [0.02380678616464138, 0.9021098613739014, 0.024562494829297066, 0.049520786851644516], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "45834c1fed1307e0ed522738ff336db886f13e3bf82bb62a5f882efdcfb1fa49:action", "state_id": "0a91e2f5e4f204fb16e834eb85fd261ad102718df5c5be7f17176d2349865040", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.53515625, 1.6796875, -1.60546875, -1.146484375], "student_probs": [0.035327550023794174, 0.8796347975730896, 0.03292889520525932, 0.05210885778069496], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "feaebdb3559fc55b0f14d97fc765b73a820595a813839c5b55c755faec7c6cf4:action", "state_id": "cc07a13bb4b41cd15edb4a85e233f2f24ec04fe0bad29df3c34003e5a0be5234", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9375, 1.65234375, -1.74609375, -1.46875], "student_probs": [0.024976670742034912, 0.9048652052879333, 0.03024553321301937, 0.03991260752081871], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a73b00106559d15ddb0a3b8b9ceb946d379abae3c95fa04edd6bff316c6fc1e6:action", "state_id": "714b25b64e6c164edd03633880aeaa8b2d125dad386d81c5b505505fe7dd641a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.859375, 1.29296875, -2.08203125, -1.34375], "student_probs": [0.037221912294626236, 0.870651125907898, 0.029792042449116707, 0.062334973365068436], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "58891298ac24ea344aac4a6f79b89cd603dd7c67d3cedfa68dbea580355e2349:action", "state_id": "26c8f37daa73f80f481ddb054ea428be39a7a2cb4fc3829889b2a2553d8e69d0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.5390625, -2.98046875, -3.15625, -2.20703125], "student_probs": [0.2796117663383484, 0.17982710897922516, 0.15083923935890198, 0.3897218704223633], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c7e8625f248b0c4fa4a292de44220a9a10ae49c5941adc46afab038e9ca7139b:action", "state_id": "fe349983761aed266fd9c82ee7ce73e6a45982c7ed8e65a7f9d58aa5dec95b4c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.48828125, -0.75048828125, -3.3359375, -3.12890625], "student_probs": [0.05249388888478279, 0.8111797571182251, 0.06113230809569359, 0.07519402354955673], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "69cbaa78b520a8e47cd753c500e61cf67bb3f289492d1f1d253f903dc18dd572:action", "state_id": "bf7296b1b85031f64b7d3e0e66a794bb87f3b1b43ffda6b959110aeef15a0146", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.96484375, 0.45703125, -2.1484375, -0.577880859375], "student_probs": [0.14443428814411163, 0.5986635088920593, 0.044222377240657806, 0.21267975866794586], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f962044563da8bcc4933fbf1e3e9272f988fce34704037abca5c1a85fd20ed3d:action", "state_id": "7bc3d1d03a5579e1109ba707d4f63e0510bb61c97a33ea86068d981057ad9ed6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.703125, -2.42578125, -1.91796875, -1.546875], "student_probs": [0.28891322016716003, 0.1402561068534851, 0.23305688798427582, 0.33777377009391785], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "156be65ea44ed0fce4e3bc08480295bbbf573703f7d29ca666e6eda77f6a8720:action", "state_id": "42196c12ae4976da620abccd062c9116158757d800002fa01cc71a732f931916", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.48828125, -0.7685546875, -3.37109375, -3.1640625], "student_probs": [0.05352329835295677, 0.8122787475585938, 0.06017786264419556, 0.07402003556489944], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0de86cdd8dbde1bdf7ae5f8c7bbb0a2ac2f4db6c29407e96e3924409072e7d68:action", "state_id": "bf26280ab568504f27165a48cd2972cfe9615f92df8da17e5e71808c68eae110", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.048828125, 0.47265625, -2.1171875, -0.6600341796875], "student_probs": [0.13517551124095917, 0.6189709901809692, 0.046442486345767975, 0.19941097497940063], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ffbc747259b66afad149cdd157a2d80b85a640c7bdb8b9a3590199ff964295ec:action", "state_id": "86b2801f26e324981dd5d6a5f930adcb0f33539b67dc0134662de86600ac4f15", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.69921875, -2.4375, -1.79296875, -1.515625], "student_probs": [0.27855363488197327, 0.13313044607639313, 0.2536259889602661, 0.3346898853778839], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "01975769be5413b290f8af25b53c3b438fca3258bfbf8a6f7f524fc5161a020c:action", "state_id": "62729ca459e0f45c3c58ea9e250aa6e39bd6e6d4ecb0332ae60ce6e6f553edb9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.97265625, 1.130859375, -0.1796875, -0.8134765625], "student_probs": [0.030796987935900688, 0.6860376596450806, 0.18500551581382751, 0.0981597751379013], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "08e06a2d9df14c2d76be81d446bde5b0065780852d764958048b532ecc393a91:action", "state_id": "19009418e29fa3c75856010d1f67d1b48462ac7bdb0575331f921c1e7e895911", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.181640625, 0.64453125, 0.1640625, 0.30859375], "student_probs": [0.06456156820058823, 0.400931715965271, 0.24797363579273224, 0.286532998085022], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "71d051781d8d0a4437d7a1d7a3c96d3730587eabe61f790dc4b3816f1534ef3e:action", "state_id": "bff475598fa6a6faec9ddbd2147d279a68fb616e44ea05c90bc4f819676e1136", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.154296875, 1.302734375, 0.1328125, 0.32421875], "student_probs": [0.04835859686136246, 0.5643503665924072, 0.17516937851905823, 0.21212159097194672], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aa51e602333790efa8c31e7ea4d27c2cf53ef8fc365e9e3f7273aae5acfd6256:action", "state_id": "802b5e6482efbca9f187912ddac8276b0c64c4d898e076c88447e876ebc74d00", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.76806640625, 1.78515625, 0.16796875, 0.48046875], "student_probs": [0.050292886793613434, 0.6461852788925171, 0.12823939323425293, 0.17528246343135834], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ab6a3f54c72697c590487ed2cceb70a52c0cf78c4bfa5a45d2c963f82cf12ad2:action", "state_id": "f55ca97493f54806f5155faf24b7652dbb435775d9d5488dd117f5799aa7365c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.97265625, -3.54296875, -0.3486328125, -0.47509765625], "student_probs": [0.21797724068164825, 0.016677793115377426, 0.4068375825881958, 0.358507364988327], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2817f5eed0dd0b3a44a50ed742fb68753008d5ad808796836568b8aa66fc4120:action", "state_id": "7522571d1dd83f121a475fccd01443d2b6c08be0ccc7e743a8559b6421b4b31e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.95703125, 1.248046875, -0.21875, -1.0546875], "student_probs": [0.029576731845736504, 0.7292860746383667, 0.1682194173336029, 0.07291772216558456], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4dfc565d851adebc73683c16026a60b8d6bf14e3abe742e14d6824597740b7dd:action", "state_id": "1a6154fdce2019e13baedbf55e231f2197311d775238912cc17e243ecfdb47a0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0126953125, 0.5078125, 0.09765625, -0.1953125], "student_probs": [0.09195791929960251, 0.4206658899784088, 0.2791314125061035, 0.2082447111606598], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9eb26916b6daf96a3cec37443aef9221bb09efdac22ac62cd33a27a9e879f0fd:action", "state_id": "0da497eb82455071dbaf2639bc7490170fe3dd4b20ae3210a9ea27c71958a610", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.86181640625, 1.21875, 0.0078125, 0.4453125], "student_probs": [0.06626652181148529, 0.530728816986084, 0.15811358392238617, 0.24489112198352814], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f85578be4a7f055325f407c32b3003aee312870df9c68d6bf6f28012750543ec:action", "state_id": "4973ff56a082ee776af989becb90fa060c7866f6ac990ed4244db5a92eb162ed", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.046875, 0.869140625, 1.08203125, 1.048828125], "student_probs": [0.11344567686319351, 0.25816264748573303, 0.3194115459918976, 0.30898019671440125], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3bea75bb1f414892e3b64a89549a264d791edca5f7b0a3238beafd7aa0088923:action", "state_id": "149619fc9f94a021fe9af53e87db396581eca58322a2dc71453023ab05b1c532", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.404296875, -2.7109375, 0.1953125, 0.05078125], "student_probs": [0.2223556935787201, 0.022145573049783707, 0.405000239610672, 0.3504984974861145], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "49a391dcdfa2dd35fa3b2348bbd07783965eba0e532fadb04c607fbc20a757aa:action", "state_id": "7e3d6f4d3587fd64015277a734773288eecf2610e4dc742566af6066ab35be4a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.9296875, -0.474609375, -4.046875, -3.078125], "student_probs": [0.07227211445569992, 0.8417781591415405, 0.02364734560251236, 0.06230245903134346], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0e00814683eab9ecd5f346022a5c9aed74e876cb50ce15fe686179fd9db6afd9:action", "state_id": "8e394d1e6abb9eb112dcacdb6f2f5f46637ff58990c2747c26a0ef23f4669779", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.36328125, -0.3701171875, -2.66015625, -1.94140625], "student_probs": [0.09428027272224426, 0.6918962597846985, 0.07006315141916275, 0.14376024901866913], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "21dea10fa4fed6dd3728a8d51d658e26e1fb31dc31510a8e8f5faadba1397d2b:action", "state_id": "6b73fb249bd2ffbcfe77bee71c4893ea593274f4b2c0e595381cb1d529e217be", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.24609375, 0.16015625, -2.25390625, -1.7578125], "student_probs": [0.0679624080657959, 0.7538584470748901, 0.06743351370096207, 0.1107456237077713], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "140a2ac2f5e1a74ca9849b9b8750792790a0a745f82c01620f442da8247dc919:action", "state_id": "a7ee4561e8f453cccbf375fcdb6473a676e86c50c229c113ea9844333a1ebdbd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.71484375, 0.23828125, -2.7109375, -1.71875], "student_probs": [0.10620100051164627, 0.7487899661064148, 0.039222076535224915, 0.10578696429729462], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f5b7097b766e674062ea429005ccbe7bfe2db1b14908ebdece7566e4d623b993:action", "state_id": "a5a785784de2a335d0b4d4367dbb2cd18b1eaa2d00d6354181ac47cc4e92048f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.57421875, -0.04296875, -2.97265625, -2.17578125], "student_probs": [0.06357249617576599, 0.7990559339523315, 0.04268055409193039, 0.0946909487247467], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3f417321389fa908e66e2f31caa76f5c0d8b21263962043628c810b4287d246a:action", "state_id": "91b1eade19377dba4c4a49492d5b8bf6760b6cf05107cdf1238ed8b7c5efb2a3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.94921875, 0.5859375, -2.30859375, -1.5625], "student_probs": [0.06333661079406738, 0.7992067933082581, 0.04421607777476311, 0.09324050694704056], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "53fb9f7035968c77dfdbfb069c079e34da8cbbabccdc066e5629454a16df6f12:action", "state_id": "ffcc2d370c6cd6c264e9582344c46331b5242857a4fc834bce0f57ec653c7bb1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1171875, 0.83984375, -2.53515625, -1.7421875], "student_probs": [0.04473444074392319, 0.860724925994873, 0.029452387243509293, 0.06508822739124298], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "17c91b095eee9183bf873d47bdb5b392dfa278bf07ab4adb681d833886d456df:action", "state_id": "50c399ea85ac084ab74c689ef351fa5b29e80bb3939f77a5f46afafe726cbbbb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.08984375, 1.734375, -2.1015625, -1.390625], "student_probs": [0.02008131518959999, 0.9196641445159912, 0.019847361370921135, 0.040407221764326096], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a1c0145927289805176dfe3e22ab036f9e52a8bff22e94fc7167298679bd0181:action", "state_id": "03d1174ca38a984d08c4093bcf0d4503625485a43e406bcc2194a10969c3f745", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.67578125, 1.58984375, -1.72265625, -1.072265625], "student_probs": [0.0333564355969429, 0.8738211393356323, 0.03182893246412277, 0.060993440449237823], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bf75b47950931f1d30016c0c826090c7d271686d97abea35773c5815a59ab124:action", "state_id": "1c62117214aa04b6278e755eb63ce05afdcca7e013cbc01e827a107292c244bb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.41796875, 1.7890625, -1.75390625, -0.58447265625], "student_probs": [0.03481694310903549, 0.8601745367050171, 0.024882545694708824, 0.08012598007917404], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0409535734d7948061a5bcff8b834560c86f6066beca6f44410a283424c7509b:action", "state_id": "f7d5a70841ba2d18c6a1e2205e113e35a99e48baab385c948896a880282346a0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.890625, 1.40625, -2.12109375, -1.29296875], "student_probs": [0.032637014985084534, 0.8821146488189697, 0.025919051840901375, 0.05932930111885071], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "25973a241fb452c9eab6241233673ef5e7df6eecbd8cdcb74280638a6f3ea27d:action", "state_id": "75fa0a6260709f6761fe698725df0fabd1874d0bfbe1b8100d762da77c0c61f4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.85546875, 0.41796875, -2.4296875, -1.37109375], "student_probs": [0.07752517610788345, 0.7529811859130859, 0.04365788772702217, 0.12583577632904053], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9a928d90d01671be616121b9ffe37ebe56e5d643b2727ecc16dc1b49c5fe43b7:action", "state_id": "76bbc96252c217a102a62cd924b37fb56cf4d901d0058ce5eb7bcc52665a4f23", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.91796875, -0.4091796875, -2.23046875, -1.28515625], "student_probs": [0.12291403859853745, 0.5557253956794739, 0.08992583304643631, 0.2314346879720688], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b16c553252c6b2cb622a40686f1baf4bed7044db09eefb6f028ec61dab528205:action", "state_id": "19524739c6a5311386e40d632c88cebb00648745da71f5dad4c0aeb133584bfa", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4765625, -1.83203125, -1.39453125, -1.0625], "student_probs": [0.232596755027771, 0.1630142331123352, 0.2524814009666443, 0.3519076406955719], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "667eab921383097ae2ae9ad579b508cf2f2522e054186f314763138f8a04601a:action", "state_id": "885f764e44cab42e991a953e6341851b6ffe278492a7786a9164b014cdb37ca0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3828125, 1.078125, -0.962890625, -1.3828125], "student_probs": [0.025187712162733078, 0.8021484613418579, 0.1041964516043663, 0.06846729665994644], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e4fc3da6ba2ba2d4506fd3f63485b86d0d8c7520ec9451d662d687d5b2b26f4a:action", "state_id": "5f13db21a37ac025c7ec6d7566bc29f483b903408ef025b36a614ce7cf34ceec", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.078125, 0.75390625, -2.00390625, -1.2578125], "student_probs": [0.04688635095953941, 0.7961263060569763, 0.050498586148023605, 0.10648871958255768], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "061634efe1cf40249a548894bb44647735887ff506f3494078c7a403c9a3b52a:action", "state_id": "42ee72eee71524bfc78c507d1db43ae270af1bea970c0ff5e9e1cead6a4f2d33", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2265625, 0.91796875, -1.2265625, 0.14453125], "student_probs": [0.06907176971435547, 0.5897374153137207, 0.06907176971435547, 0.27211910486221313], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c6375b3d4b7df82acd88139bde79787e5913303d486b88953c3b57593784dcc4:action", "state_id": "0953d71c426847e9e4121a063fcc588a7bc49ea9466eda98a85e5873b28311e2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.98046875, 0.765625, -1.9140625, -1.19921875], "student_probs": [0.05041717737913132, 0.7855827212333679, 0.05387886241078377, 0.11012125015258789], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cf408acd0361b30cc91e574ccb2297766bdb28ba849793a1fafc4fedd9e20baf:action", "state_id": "6b4c0f48e3438e0f8f2aacbd5df6653b1523b218ea5f31d86c2abeb23efebf8f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.91015625, 1.234375, -1.3203125, -1.48046875], "student_probs": [0.03629859536886215, 0.8424465656280518, 0.06547201424837112, 0.055782850831747055], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6c315750fba2660536ec92eeabe80d33665996c8e4438382d6109e99435f9ad0:action", "state_id": "ea3d58a8fcd855711b8580d55616c44adeb01f204c35a48d7a8198a4e08f6969", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6953125, 1.02734375, -1.142578125, -1.25390625], "student_probs": [0.051246337592601776, 0.7800050377845764, 0.08906607329845428, 0.07968252897262573], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "19947f51146db769da3177cfc6d2cbe6a0f770b2505ff138cf7eeca6f5d54639:action", "state_id": "4ccb1d9b37660af2a94a65858aa9999f480501629bdb8bd89d66faff9c3d5393", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.255859375, 1.337890625, -1.05859375, -1.1171875], "student_probs": [0.05971338227391243, 0.7989562153816223, 0.07273492962121964, 0.06859557330608368], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "67a12aa699870b7287851f1856760cd3204a60d03b1e5632c1e6ada08e07bcc5:action", "state_id": "f7e6e96fcdf208f03b8791a46dd6cea48cbc823fdbeb565fcd87374969ece295", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.583984375, 1.720703125, -1.060546875, -0.2578125], "student_probs": [0.07676002383232117, 0.7692157626152039, 0.04766138270497322, 0.10636280477046967], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ef20a83d56538820e08655bd84ae178b250dee9a2132c7509dec33381202a847:action", "state_id": "f0fe41ef06907c5cbd87c0aaa7254aed8aae62ce116bdb0216dced5960f6f9c1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.21875, 1.470703125, -1.177734375, -1.072265625], "student_probs": [0.05579346418380737, 0.8214818835258484, 0.0581294521689415, 0.06459525972604752], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "97637a792681ee93802c40f685ab8af7ef7f7a6e8c245baf9702663f086aa058:action", "state_id": "dfe2d0c151ef04ccc82c6bfab68ad144ca761e09fe72cb360dcac72578231668", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6953125, 1.51953125, -1.578125, -1.14453125], "student_probs": [0.03477252274751663, 0.8658149838447571, 0.03909579664468765, 0.06031668558716774], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "15491249977bd1db762f52a0fc5ed5a695420d843a90484fa96e7974b8095c99:action", "state_id": "3fa823bd2ec37a5730e408711fe5612e6ff2a39fd564a4eae01a3a58f8d66b65", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5859375, 1.939453125, -1.23046875, -0.9189453125], "student_probs": [0.026080885902047157, 0.8858904838562012, 0.03721349686384201, 0.050815168768167496], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7aee8752ff5cefd3b5dfecd2db0ea1b370b6cd1d8ae2beac574e7364fe3b3c47:action", "state_id": "9ac786d38d12bfc8d95aa3d233114a7b9e419ead3172dcadb8b5abe99c42bce1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.51953125, 1.3515625, -0.9150390625, -0.758056640625], "student_probs": [0.04419289156794548, 0.7802838087081909, 0.08088724315166473, 0.09463606029748917], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "126692171a0d4979215c1f14f5766803c474f6c1d74d8f39476bc4633161f303:action", "state_id": "b043f3e30b1e00a6ad2fed2be9b05f4b5b5dd4862cbbc3109c91982e89695b81", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.45703125, 1.28125, -1.71875, -1.4765625], "student_probs": [0.02092762105166912, 0.8794978260993958, 0.04378761723637581, 0.055786870419979095], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0cf8214cd6fc79d24f69b277462b08789023dc7a8d3708d8f043882b087e317b:action", "state_id": "7b41685a029ad9b765728e90681f49231a3e4bb2b1ceb67fbb1f74bdbdeb3d3d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.99609375, 1.43359375, -1.77734375, -1.21875], "student_probs": [0.028338884934782982, 0.8747362494468689, 0.03526831418275833, 0.06165649741888046], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5c62f0df2a89abde45838b4a85398b756c115254e298e92d82ffba245e2c1fa7:action", "state_id": "86453fc410bf807c80c7ff659ac4d1fb96db2d3d8a2e85afcbe1dbf29487558e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.24609375, 1.3359375, -1.6875, -1.4140625], "student_probs": [0.024394599720835686, 0.8769000172615051, 0.04264694079756737, 0.05605833977460861], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e71ce0deb98c8981ce4898ad932f13f0e564204e8e92d67542dbc25e3743fc16:action", "state_id": "b7966d72e524d3550d3a21bad0241f5a290de87593f5b124b190104a007a2c5e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4765625, 0.8671875, -1.74609375, -1.087890625], "student_probs": [0.07321181893348694, 0.7628846764564514, 0.055914606899023056, 0.10798893123865128], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "55d542ce5d64358aebf3491f34bb6c704f722d217aa59f9598408b26df37705c:action", "state_id": "01059c35e8459bf4bf8994d524e6be40a0d0e8ffc4762769777af6e672dbb2b3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1640625, -2.953125, -2.75, -1.796875], "student_probs": [0.28947558999061584, 0.13150019943714142, 0.16111741960048676, 0.417906790971756], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a6077eb2082950586a56adbb9513b3b38aea1fba08f0fad6bedb58409d1849e9:action", "state_id": "47805ca750dab8f0079d9e8428db53a447f7a424fc0f88204530aa1ceab44202", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.609375, -0.615234375, 0.546875, -0.59130859375], "student_probs": [0.1615409106016159, 0.16059716045856476, 0.513375997543335, 0.16448591649532318], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "57ba9050f48adff313fdf20a12ab7ca350ebb14ccb538dbebf6abcbee4558e99:action", "state_id": "da1c4bfb23d97937970c7ea204e138cac1574acfbf01fac831cbedda3e01aeee", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.6171875, 0.1328125, 1.28515625, 1.078125], "student_probs": [0.19410261511802673, 0.11958315968513489, 0.3785528838634491, 0.3077613115310669], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4a488d05939c7067b8d61de21f06fa8090a5f811290ceaf83756aa99c4a496f5:action", "state_id": "7f251a9e11e46b3ca1eff56560ed70775fe1695c97f7e7649bee7ca6318b9cd8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.1484375, 1.443359375, 0.79296875, 0.87109375], "student_probs": [0.08890432119369507, 0.43674853444099426, 0.22791367769241333, 0.24643345177173615], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "95d82fe03b3999759a78d39767a031be8a9fafd7d81437d588a4215044d8386a:action", "state_id": "8a07cd643cf95c3e5120811b99d03185c7041eb183d8ddf6377a7aa69280a090", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.38671875, -2.39453125, 0.25, 0.0], "student_probs": [0.22238567471504211, 0.0298624150454998, 0.42036858201026917, 0.3273833692073822], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b05a58b1f89efeb744bb5f44560e8555dabdeb8d80898989f062e2b10a803f71:action", "state_id": "0c86cffcc4ca018d414e1f0c3ab6f3750f3de51f4c5439b70086dd0853e9686c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.998046875, -0.8828125, 0.642578125, -0.06640625], "student_probs": [0.10184130072593689, 0.11427982896566391, 0.5253373980522156, 0.2585415244102478], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "77ea2bf58228a97819d50eaed88c554a5c1816a07ee6cbc0d117c844f4264d57:action", "state_id": "d93f837826a37c78a9a76588a8708b0e932bc8f7d8fabae3b07b0372d5629466", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.00390625, -0.447265625, 0.8984375, 0.732421875], "student_probs": [0.1624676138162613, 0.10347259789705276, 0.3974264860153198, 0.3366333544254303], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "af735fb7157b3a163b2ca1ce9ec0b27ac5cf8c4091640c8a03441aec08ed6a72:action", "state_id": "49731ab8f3af9edd0207919caae68016eda66fdbd0d12feaedf5ddd6a7e2f4db", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3271484375, -3.0078125, 0.0625, -0.03515625], "student_probs": [0.2574617862701416, 0.017640672624111176, 0.38013243675231934, 0.34476515650749207], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "be4b17906370f3e4f7eea259643c2b89a93ec7a38af1c72d09ebf8d1e60b8267:action", "state_id": "b23e582e00bf34a698dd7f86a50c063a3c46f458dd9f24a5138248f788196f31", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.66015625, 1.1875, 0.09375, -0.548828125], "student_probs": [0.03695105388760567, 0.6373063325881958, 0.21347089111804962, 0.1122717559337616], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c7bb0adc30b92c1790e9b5558ab3558dd51c6e827dfecebb86709affeff4825d:action", "state_id": "91c25f5687cfa628eb690557909788e2164a8a94a4db5553a8295c77e7dced07", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.13671875, 0.697265625, 0.1015625, 0.34765625], "student_probs": [0.06613467633724213, 0.4139220118522644, 0.22814340889453888, 0.2917998731136322], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7f019df2c347de103a1b06fb4d04dc391e87e52e6e998cf3c820caa2c97e58ac:action", "state_id": "aad383de425aa71c0ad38f5022d83c3272eeb0937585baff3d6b068a8f5887ed", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.18359375, 1.490234375, 0.0390625, 0.38671875], "student_probs": [0.04219462350010872, 0.6116259694099426, 0.14330124855041504, 0.20287810266017914], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c2f12d5fc981ca34f1c560c7e2ece8e129216b0bc76173f8a9d92a34a64cd9ca:action", "state_id": "3fbf3068a6b249a36be2f0284d8f68ed126e8f23c917d037241623fe0e977324", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.359375, 1.91015625, -0.54296875, -0.189453125], "student_probs": [0.030503520742058754, 0.8022123575210571, 0.06900978833436966, 0.09827443957328796], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "023b122e994c12c5104ab83ef2994e6e1a8178374ea49f72f10cd06996ad5f25:action", "state_id": "85f028efe7ef428c1592b4e14c934f22ba066891ee1a5a2da2a83ce4fc2f8a60", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.421875, 1.92578125, -0.4169921875, -0.265625], "student_probs": [0.02829207479953766, 0.8045136332511902, 0.07728226482868195, 0.08991201967000961], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7a4b3dce8828f4020b6197acebde33b20dbb1aa9e94fe663c8ce0def74dede9e:action", "state_id": "0e6ac9f197434ff6ac819373a1ef93d25961dc8bb325a98379a83cf3a610fbf0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.31640625, 2.169921875, -0.32421875, 0.06640625], "student_probs": [0.02478375844657421, 0.8095808625221252, 0.06684496998786926, 0.09879045933485031], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a2915f3af8d8aebc94281fc6c7026d7ed601eece085d1fdcd4b51b84cc4c468a:action", "state_id": "572817b50614d4ac5fbc85802332fd89fda3064d0de23d2f71a9839ecd2cf43c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2890625, 1.951171875, -0.4443359375, 0.0], "student_probs": [0.030772605910897255, 0.7859233617782593, 0.07161836326122284, 0.11168555915355682], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b7b2744a48dce12e93c81e2d00eeef98dccfbf2e3d795bbdd7284c015b72c549:action", "state_id": "7e1a30760c05c833405a12f1e0ba1396409d8fc78a9d1e225351198629c68cb1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.140625, 2.01171875, -0.541015625, 0.03515625], "student_probs": [0.033952511847019196, 0.7941771149635315, 0.061841342598199844, 0.11002900451421738], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ba5d8ed0f93512d0e5083315d4a93902a764a56b1718e76ef681b5f9b506b75b:action", "state_id": "8a22168b0fd45685c04dfd26a7925d7ed85c62a6b19fb40e6a0d63bc4e8381bd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.154296875, 1.681640625, -0.7783203125, -0.416015625], "student_probs": [0.046306755393743515, 0.7893622517585754, 0.06744176149368286, 0.09688930213451385], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "82c2a072abf1127aff6e04fba7a3cd42a7f0c77dcd8c70fb2923c12a07fbf487:action", "state_id": "8612844db250dbcd2f1b97ef7c9339618856183c85573a22820962f35f6f5b2d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.068359375, 1.783203125, -0.595703125, -0.265625], "student_probs": [0.04514530301094055, 0.781682550907135, 0.07242434471845627, 0.10074782371520996], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "96915b26f14507633dd71da7cad1a58469d368f861a7763957ce602b32a18543:action", "state_id": "54424f66601f57c74f87c0f9ca2da643a87b44ab9869049757f71d46aba91a7a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8837890625, 2.01953125, -0.40234375, 0.05859375], "student_probs": [0.042700208723545074, 0.7786207795143127, 0.06910651922225952, 0.10957252234220505], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b6f1686df86dbb3e898d583d8613bd1dec2bc69d7772a7cdd08675470e58542e:action", "state_id": "93e6c45354782c5dabe8d14209c69c4a8655c57ca8d22a172e4eeab06c71bf92", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.970703125, 1.787109375, -0.203125, 0.05078125], "student_probs": [0.04608894884586334, 0.7266069650650024, 0.0993005707859993, 0.12800349295139313], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "27977375e08b4c6b7ba95d2f68ace5698f4af9dff65740b8520190856752d89a:action", "state_id": "07218144b9abc1b606dc3556eaab5986d535e3265e649209da21e2f9518627c5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.17578125, -3.90625, -0.6256103515625, -0.60931396484375], "student_probs": [0.21926124393939972, 0.014293361455202103, 0.38010019063949585, 0.38634517788887024], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "932ab4156ad1549045e00eda88d793ac713fc9c24181f2894704a003bc35d0ab:action", "state_id": "4d8f7f5a759b42f478aac8d0513d5b6dcf587fc9a32d3e2d35b771213ced1fc0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8671875, -1.5625, 0.15625, -0.80712890625], "student_probs": [0.18714110553264618, 0.09336815774440765, 0.5207657814025879, 0.1987248957157135], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1f94daf37bd3e14bf67018a906484a45c3c47758d95c64ead4b6c71e76190279:action", "state_id": "ecfb88224a4a9b87a1d84dfb4f1a49af5fde5406821d7433da95d2ea2911a76a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.869140625, -0.0859375, 1.064453125, 1.162109375], "student_probs": [0.25375083088874817, 0.09763876348733902, 0.3084825277328491, 0.3401278257369995], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "05e11a1c435967028fc1570b3c56d72de3366790fb9c25184c4913e229de4167:action", "state_id": "458ba46dda133fef864e8be2f79e790b8be92e976d2834bb18920198c8e166b4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.08203125, 1.240234375, 0.59375, 0.830078125], "student_probs": [0.10861244052648544, 0.4075043499469757, 0.21348513662815094, 0.2703981101512909], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2e46d0031f97a0f4a0b95349a10b49e9a467545c14910eb3542e68d96802c6fc:action", "state_id": "644ae19b50288ee8e33fea3715c552a98d59ecf1fb04f202331d8f915b55e11e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.32421875, -1.71875, 0.33203125, 0.140625], "student_probs": [0.20976386964321136, 0.05201078951358795, 0.40433043241500854, 0.33389487862586975], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "da794b2782a1568f40d95e56fb05320128ce23f40111c4e3ab8a2a54f8e2d816:action", "state_id": "5b58d2a1ed0dadc964f6f22c41b4e5bc09ef33e57d4c19bec3291990baad5ffb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1171875, 0.9765625, -0.09375, -1.08203125], "student_probs": [0.029904775321483612, 0.6596887707710266, 0.2262081652879715, 0.08419827371835709], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "753c6a2d79ad8a068448e2741c270e1092f0584c714805858bc7bfdb364319ff:action", "state_id": "f093da49bd11c0e46a325c925d1e37e9c8ada1a159b01c6b96890cac1dd20fc9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.640625, 0.81640625, -0.244140625, -0.296875], "student_probs": [0.04867488890886307, 0.5680415034294128, 0.19669367372989655, 0.18658992648124695], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "357e95f2798e082b1b4511b848d3ac46800d86d82a82cacfeaea12723b4469bc:action", "state_id": "43d2d6539dfe3f6e53d6ed57036019792e491fd1d0f2830ad83168540fc2a365", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5078125, 1.2265625, -0.376953125, 0.10546875], "student_probs": [0.04078688472509384, 0.6281226873397827, 0.12637072801589966, 0.20471970736980438], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9736786d8d78ae056a11aa485dffbda4705c6d6013a2d2ff0a804d2c34482550:action", "state_id": "d65583fe465e21f334ad8f93e8a1000df2cea365da2aba8705f637851260b517", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0625, 1.7109375, -0.10546875, 0.53515625], "student_probs": [0.040718309581279755, 0.6520461440086365, 0.10602861642837524, 0.2012069672346115], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "192213f0cc8cadfdd28f2a868f75a736a8e612104d854720c5774b13f451e2ff:action", "state_id": "22667dc5dab6cd13acf261dce664e59859de165a659d2b6151b2a191bb16ce23", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8134765625, -3.375, -0.236328125, -0.328125], "student_probs": [0.22307057678699493, 0.01721816323697567, 0.39727815985679626, 0.3624330759048462], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c9268f188f48dacacd7fe939653336f193285df9529b6e023eae8e82e676c021:action", "state_id": "a33ee57b47973fc5a434824b8d587cac170fe02a9884b757abff6431ab186214", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.1953125, 0.20703125, -3.00390625, -2.92578125], "student_probs": [0.029802074655890465, 0.895087718963623, 0.036088861525058746, 0.039021361619234085], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "656f616dc0fcef14dbfbc9859c2256fde19ee2e1a4c3bcc55dde2cf96d16d05a:action", "state_id": "508291ff89fc40d53336327e4f74e5041da0a67dd30caa7454fd68df923c0a17", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.69140625, -0.0703125, -2.37890625, -1.25], "student_probs": [0.123208187520504, 0.623263418674469, 0.06195296719670296, 0.19157545268535614], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5f890697c432f7bc15d58ef3822b9c51abe35606289f8a70e93ab7be40f63f62:action", "state_id": "710902c293f976166d5350e49e17e52533bb5f13d6a685b986507f6fa3087796", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.01953125, -2.76953125, -2.24609375, -1.70703125], "student_probs": [0.2749916613101959, 0.1298968642950058, 0.21924248337745667, 0.3758690357208252], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aab207c747998cd014e8cf4e29aaf7e69eaab47fb1005cb190b5c973a7f6e162:action", "state_id": "6ca67cf1a8e304fe1d86c0376bed600c168836929061eb8bb6b866acbf7b0d89", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.41796875, 1.17578125, 0.24609375, -0.509765625], "student_probs": [0.04516622796654701, 0.6043174862861633, 0.23851023614406586, 0.1120060384273529], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2f2c2221bc17429805a863127a636c1c59ff948bfdae747bb6b76a0a907bf0bf:action", "state_id": "a80376c09464f1bd71ffbd4b81ae9128cd97460590be019183cd7609578c8e20", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5927734375, 0.55078125, 0.5234375, 0.359375], "student_probs": [0.10222402215003967, 0.3207690417766571, 0.3121168315410614, 0.2648901045322418], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4f4744271308b030bcc47457d5a6ae539883b80c3fecb82da2725f30dfe90dcd:action", "state_id": "c89c78a6659ad56c1a5228db2b331c3f1d4f306abb05d4c85a72d9dd6082f420", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.177734375, 1.330078125, -0.09765625, 0.3359375], "student_probs": [0.048154860734939575, 0.5912474393844604, 0.14181171357631683, 0.21878597140312195], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1b1514d4ecb3da6dcefe9175dad8e60508b76ea95f141b0894673bc82b8d2156:action", "state_id": "fe8df1592782c36e07d87d940d11c5e09b634f75c5ea700aa2d089397e8ba5d7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.31640625, 1.69921875, -0.404296875, -0.06640625], "student_probs": [0.03652067482471466, 0.7450888752937317, 0.09092071652412415, 0.12746967375278473], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2513eb0c1dfbff6d5088c7ef31344e8295d406054075fed58b017003ba6bda64:action", "state_id": "6e9ab4fd068796b98ccbf1361df120a393fbf947c8156b8a79bf4ca917cea0cb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.515625, 1.8828125, -0.56494140625, -0.4541015625], "student_probs": [0.02747587487101555, 0.8220044374465942, 0.07109321653842926, 0.07942647486925125], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d5150b7f2d608ab8ee58b817806e338eb69cb8368d3fd4f5ff46dcc8128b35ce:action", "state_id": "bf545b7e629373e61a241d8e6133f9d5153b4d976b9cff4ae3c19c0746e250c2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7734375, 2.1328125, -0.8388671875, -0.72021484375], "student_probs": [0.017817314714193344, 0.8857375979423523, 0.04536500945687294, 0.051080018281936646], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1fc54da8d01ed3eebbb549e8f8fe6585ec1c8fecd3653f649b00b735c741de7f:action", "state_id": "95b27f0e459cb49375863fe9762d246d6e0d58699a380088bb80a268db20c4b0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.71875, 1.89453125, -0.7890625, -0.30078125], "student_probs": [0.02234637551009655, 0.828772246837616, 0.056619398295879364, 0.09226204454898834], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e3e0c65d010311fbe58751d3702df3b88ca833e1008ceb427d38cdc7cf6df617:action", "state_id": "ad28352536234fca21785b299fad8316c957e773b4f96ca55463e567c9f0eea0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4296875, 1.880859375, -0.61279296875, -0.212890625], "student_probs": [0.029377272352576256, 0.8049403429031372, 0.06649427115917206, 0.09918811172246933], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b4f46e5de964f4ca231853258701084e16267d4070f09e90a02553ceed1a71c1:action", "state_id": "95ef2512652ef9f45392ac8d64892a323bf1a3101c8ad98aefcfb6ed2c3a81e3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.97265625, 1.771484375, -0.720947265625, -0.37890625], "student_probs": [0.050895169377326965, 0.7914831638336182, 0.06546247005462646, 0.09215924143791199], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e4dc38925db46225a288dc77d33163be359b11fbeb5dfdb43852d66eba5cde8d:action", "state_id": "e582d0e3e9417cc9af4a1c1ed3695e6518b679693516de7963a63f90595d1b21", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.064453125, 1.880859375, -0.61962890625, -0.265625], "student_probs": [0.042017240077257156, 0.7990252375602722, 0.06555596739053726, 0.09340157359838486], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6459b35dc459b6829eac4af46c02e66de6ac2dbfb1bfc75132ea108fc6063f64:action", "state_id": "0a2c326024a4c0791f4f33df8fbb710ce3b8e5213f7ae62e175dd5b6a56a7839", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.978515625, 1.998046875, -0.447265625, -0.05859375], "student_probs": [0.04027320444583893, 0.7901705503463745, 0.06850703060626984, 0.1010492667555809], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e726cadc7727b5216ef495caa5f83603a29ba060049c9b900ca54b1ac3232f48:action", "state_id": "a3ed3d90a74218c1abdf342a3440bb478a4d9f72b62e75e2bb7f824f56650ca1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.068359375, 1.7578125, -0.244140625, 0.03125], "student_probs": [0.04317079484462738, 0.728753924369812, 0.09843368083238602, 0.129641592502594], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7ec0dd0cf554241cce62617e820a4f7b9d28e552d55b2fd6dedd94c2df64ac7e:action", "state_id": "aac12593cdc40fea2322e011bbd8a6e84ddbe949adc14a8be6ab9bf05ec47046", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9482421875, 1.94140625, 0.06640625, 0.1875], "student_probs": [0.04022710397839546, 0.7235643267631531, 0.11096218973398209, 0.12524642050266266], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0fdb3c6c417c1b9b2fde4608d7a20192f38b458f4c305b5b7c5732cbdee317c1:action", "state_id": "e9a25352a881fdfd3bd0bbca6a3e60a736b7b31865da66193649780bff84b744", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.826171875, 1.998046875, 0.48046875, 0.55859375], "student_probs": [0.03916130214929581, 0.6597809195518494, 0.14465183019638062, 0.15640592575073242], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "71194981f2037e9ab48887073e8f90ee4f48bae328cf22abe29360eefce2bbe0:action", "state_id": "1292ded5e0350a04d77e6e07b9e0c16736cdc238d5ed1a8978c0093a6b6490b0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.710205078125, 2.015625, 0.6640625, 0.69140625], "student_probs": [0.04118106886744499, 0.628797173500061, 0.16275504231452942, 0.16726677119731903], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ebcec37d6b95773d97b98125ee10faa6a448d2ce1dec62412445850d394b7711:action", "state_id": "dfbbadf2a7458ea369fb2d60f0438211c66f3fb2a25d0e5f1efda35022f53ed6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.28515625, -2.5703125, 0.14453125, 0.17578125], "student_probs": [0.23673708736896515, 0.024089928716421127, 0.3638121485710144, 0.37536078691482544], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dca3f8ff53f0ab9a9d9bcb853c0c572534c53fda8f0510b687a8313e5c4ea310:action", "state_id": "f8969692f31d00f5659a690056e05f49d1ec5c399cb4cc6dcbb6d4be0a966159", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37890625, 1.162109375, 0.29296875, -0.546875], "student_probs": [0.04692043736577034, 0.5955402255058289, 0.24971699714660645, 0.10782231390476227], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e24b90278e69a19f0aa8a41dddb4ed7e2521269d04ce813ff9f19770cd43d25e:action", "state_id": "7699f1b4e30c3b25a30ec7b3855d5054230dc53e648ceca1b5e6147bfed04d7a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.64013671875, 0.5625, 0.47265625, 0.265625], "student_probs": [0.1015687957406044, 0.3381105959415436, 0.3090580999851227, 0.25126245617866516], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b58d67f341cd37ca5e3985b404c6e1c1b2488a4657cf19773855c9095cdb37fe:action", "state_id": "7d508688285c8d0b6c0783f89956e0cdc58ce83e40dedccdb1cce888482acaa2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.052734375, 1.29296875, 0.171875, 0.37890625], "student_probs": [0.05255134403705597, 0.5486681461334229, 0.17882363498210907, 0.21995683014392853], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6f0705a7b92aff06c21dba5e9bdd9dff199116e1b46c39b538a2fdbc154a751a:action", "state_id": "2564e7619295dac0c122230d50227de6adb44584b15dacfbed696e53d98cf1a8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2890625, 1.65234375, -0.25390625, -0.03125], "student_probs": [0.03805793449282646, 0.7209110856056213, 0.10715386271476746, 0.13387708365917206], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d735bf904c52a24548ee9e4dbc6f4f805d1080ffca01b9ce2b0c160b40a904ca:action", "state_id": "d1baf44639423b0e00b38d71edc483d501d91c50cfbfdfd8493bc47ffb9c6983", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.609375, 1.8125, -0.80419921875, -0.5244140625], "student_probs": [0.02715679071843624, 0.8317252397537231, 0.060752175748348236, 0.08036574721336365], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "09d575d527a8096e2f1cb9926c1d27a421953ae2ebfc827c27e91f3bbb791fb1:action", "state_id": "ee391b171bcf4e351cc4c761666b753f89a667d46f2f27f6e4723f38fd1709bb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5703125, 2.138671875, -0.699462890625, -0.296875], "student_probs": [0.02093171887099743, 0.8542723059654236, 0.050004612654447556, 0.07479141652584076], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "abc09286d244a18e1aa1082098fe4e9a26e04c1fad3ec7b5394ccaf2f27967e5:action", "state_id": "474e23c89ac4bb87edaaf90f7b66e9c44875004fdb583ca7b1758822b77dc835", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7421875, 1.84375, -1.0390625, -0.74072265625], "student_probs": [0.02390657551586628, 0.8627207279205322, 0.0482926107943058, 0.06508006900548935], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "338182b0cc6b422fa5faa30dd10e7729c7d9179fcca7e227ff8362f4b3386c83:action", "state_id": "bb52b58166490b11f503bd04946e00b58debdd2718b5e458ec06e94c207ca569", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.71484375, 1.603515625, -0.95703125, -0.94921875], "student_probs": [0.030396105721592903, 0.8393886685371399, 0.0648532435297966, 0.06536189466714859], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "34c907c0a574172cef07de1f80dc5a29c87a109b2f4d667faefeed72c5860f05:action", "state_id": "8dd050217a27dbb3f23057911aed10ee8e177c6cec677efcd8b3188af9b6027a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55078125, 1.515625, -1.177734375, -1.048828125], "student_probs": [0.03911028057336807, 0.8394875526428223, 0.05679408833384514, 0.06460801512002945], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "14fb05d0204f922fbaf133fac0eaf8e8076c0a43d00b5d13b5687d7e1050107d:action", "state_id": "a9b22315086164fdec0a18fe08ab67f989a6aa55dead8627eba6886f121b7905", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.140625, 1.78125, -0.8505859375, -0.59912109375], "student_probs": [0.04418687894940376, 0.8208191990852356, 0.05905486270785332, 0.0759391114115715], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ed0427105efd2b0f71666299a18cafd54e28c270f2a521e664cf69b790242946:action", "state_id": "ad1952b7da13090cb25904ab7c88b0a216d1a94faef1c5136d28cdd981ec27ad", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1015625, 1.9609375, -0.5361328125, -0.50390625], "student_probs": [0.03852229192852974, 0.8236428499221802, 0.06780708581209183, 0.07002786546945572], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "723db174437b16c27d56508062fae81d38afbd3741e403ecc97fb5d916f5f158:action", "state_id": "ff8eedefd4ab2f80184f053c52c6e4a16ecfd0dc77c802faa4efa316de4ef261", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.25390625, 1.576171875, -0.724365234375, -0.478515625], "student_probs": [0.045837122946977615, 0.7767918705940247, 0.07783843576908112, 0.09953256696462631], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a4bcd3a73d67f4c9f25efeeaca0c6dc4a798558d6da81419fb6d18f578e34bc1:action", "state_id": "8d7aada06ec5f44c31b6c7e117855d69270d430e45fd5e66ed5b9ce33533340f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.296875, 1.794921875, -0.56103515625, -0.248046875], "student_probs": [0.03576775640249252, 0.7874845266342163, 0.0746556892991066, 0.10209206491708755], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c7d0215ee29f66c9989e1f0d2230f01a7a7ec9b537b8673d22c7303a3ee1b6eb:action", "state_id": "cecd6f3ed39cd294e6917897cc0fbaff01f000342d004e86d5ffb5b122638ac4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.083984375, 1.919921875, -0.208984375, 0.3125], "student_probs": [0.03622664138674736, 0.7304794788360596, 0.08690319955348969, 0.14639073610305786], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "23936c4b8c50a8d23ca46f20c33c4bf9afad92c2699fd2e1ed23b1775eea2798:action", "state_id": "390c4f88c47be7d71d098cb1a9df7aca604cdadabdff337b556e8b9b8a1eb3ab", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.162109375, 1.828125, 0.2421875, 0.3515625], "student_probs": [0.03389096260070801, 0.6741029024124146, 0.13802644610404968, 0.15397962927818298], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "959ced85efbdd14cf86c141f27f422665519cd91a14189652ae4a037841ab522:action", "state_id": "6adf1fe22da2066d072c3c2d3e00e84c98467ae5917c4c07ecf2cc3b2c628fb4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.47265625, 2.0234375, 0.771484375, 0.9453125], "student_probs": [0.04823071137070656, 0.585279643535614, 0.16735823452472687, 0.19913135468959808], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aa108e5543bfe1ec353a793f63caeffaadb098ef10a65f9a335071ad038dff3c:action", "state_id": "31a5b9e72e9b3d24dac4d4911a8c66a3a4782a7268e5e53fb2d11952d2c47143", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.19921875, 2.0048828125, 0.75390625, 1.02734375], "student_probs": [0.062245581299066544, 0.5640760660171509, 0.1614527553319931, 0.21222563087940216], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "878193c0fb4f1a2e609157ecb758333e37857e5bac9d0b84171dcba3e99c81be:action", "state_id": "60fe7780381d7e473600716c7f896e74b197f83ec44ecdf000b4c14105f1f69c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3671875, 0.171875, 1.51171875, 1.5234375], "student_probs": [0.12282688170671463, 0.10103464871644974, 0.3857954144477844, 0.39034304022789], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e610962f4cb9e0339523d040e1af289247c958531e495330113c96d03b049f75:action", "state_id": "54fb04dbdacda921cb919d023552ee734b33a8e22c95dc4755869d3b70aef346", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.23046875, -2.69921875, 0.296875, 0.3203125], "student_probs": [0.22154657542705536, 0.018762923777103424, 0.37539416551589966, 0.38429638743400574], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5fb973086528a1c66f8c4640c45ca29a8f82f6ffc360cabb3669981e78f1dd28:action", "state_id": "5e50a3ea41626f76df6dd34cdfb871ccb093750556490d8e99e5bded44054918", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4921875, 1.232421875, 0.25390625, -0.3701171875], "student_probs": [0.03991406410932541, 0.60870760679245, 0.22879408299922943, 0.12258430570363998], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "857c2a6d0f447f58a5fca309341f5048a43c4dadd83ccd5f0a8dde5be18a5a87:action", "state_id": "83eeec9501cab936700fb3523d568bbcef98db21e7c4de2fd1ab34d4671da813", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.94921875, 0.73828125, 0.30078125, 0.48828125], "student_probs": [0.07088956981897354, 0.38322538137435913, 0.24742890894412994, 0.29845622181892395], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a5ee7784610c53e27d045753d1517282574dc8d922ef50e563b939dda7d35ed9:action", "state_id": "d21edd81cc68be8916040590b8737f02fc3a0aea0b73bd9e28df0f5bd0cf8116", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.140625, 1.474609375, -0.125, 0.3515625], "student_probs": [0.045707352459430695, 0.6248387098312378, 0.1262020468711853, 0.20325201749801636], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dcbe5ea93d72f6cf7c0c9d12cf269bb12825feabc9bbc3bfda86e19e9d90eaa1:action", "state_id": "81504b452e1569c7b3da29bd6f080a3880c9a6164177e8c76bcd0ae305bd12b1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58203125, 1.8515625, -0.6873779296875, -0.48828125], "student_probs": [0.026723861694335938, 0.8281137943267822, 0.06537958979606628, 0.07978270947933197], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cd3a2545428e7225b61f80ca1d57bf61f93568d9c2396edfd703ef59908fcd8d:action", "state_id": "c70569b5c023cd7f63c514cf6586bc15d21aa2d0b829f8c7cfdbe0e0922c875a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.828125, 1.947265625, -0.896484375, -0.9296875], "student_probs": [0.020157644525170326, 0.8791663646697998, 0.0511736199259758, 0.04950239509344101], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2b4bf787aa2338f1193fe5f22c4d1c55777cd23cc72d5dcad6c8c7c863934c52:action", "state_id": "c5d0df0aef0725cb0f20e6de1a15ca8eda8ec6c9602e16ceffcc71c391e67540", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78515625, 1.681640625, -0.63653564453125, -1.119140625], "student_probs": [0.026223130524158478, 0.8400309681892395, 0.08270354568958282, 0.05104244127869606], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "02a831007bbb6b6886a4ea7cd5bc45e6d57efd5c9332cd262bc5e65c7b165b5c:action", "state_id": "b46dad3d9bd51929eb5692fb5b7f69808ab85e04f7a75eaf454b569fa4b4a38f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6484375, 1.455078125, -0.7900390625, -0.75341796875], "student_probs": [0.035608869045972824, 0.7932277321815491, 0.08401481807231903, 0.08714856207370758], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c21671ec2a1453fedb46684161ffdd813b0d5c3048ea9e496a3c3035908a5e45:action", "state_id": "49b6ff68a8cb994b20e8b055577f571c03913e6e3aa58127a51a3067262b6d1d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.943359375, 1.794921875, -0.65283203125, -0.0546875], "student_probs": [0.04943295940756798, 0.764252781867981, 0.0660984143614769, 0.12021588534116745], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "429ef4569d0f31a0486500999e34ab425f500f298fa2541a29fd89770c71a6a7:action", "state_id": "bd57ccb44b5e2246080d59ca64a11e345b44eeabd4d80d83c1f68ff59c3b28a1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1953125, 1.697265625, -0.916015625, -0.57080078125], "student_probs": [0.044985681772232056, 0.8115308880805969, 0.05948006734251976, 0.08400329947471619], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b0b05fae13b8ad575009c9890ed80b8d3fb00a3395945de0234bb220d0e066f8:action", "state_id": "10c29a82465484f8451eba05084547b8711b830103c33b8bffcac3766a7de020", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.06640625, 1.783203125, -0.7042236328125, -0.41796875], "student_probs": [0.04623199254274368, 0.7989364862442017, 0.06641046702861786, 0.08842100203037262], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ced11266668a1cdb4e453543658576c0530c1bf28da476686450143d0a2847cb:action", "state_id": "2a86852acb356959d63b2243ff9e133336512596aaa995cd252af9d3f8370d2d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1015625, 1.9140625, -0.5654296875, -0.21484375], "student_probs": [0.03915676102042198, 0.7988699078559875, 0.06693392992019653, 0.09503942728042603], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cd6e3184c63bf6d2b42838b84cbc6c25fa9ec46ba54f40b8933877db0080993b:action", "state_id": "7819e94feb9ad79980ff1b4b1ea20d489948b3884b1c9787380a70d9975f897e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.27734375, 1.66796875, -0.4169921875, -0.2890625], "student_probs": [0.039892759174108505, 0.758624792098999, 0.09430614858865738, 0.10717640817165375], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b4fc4bdbc76fd2ef99597aeed819f6713e85184fca00f40a1222b7497ef382db:action", "state_id": "3da736e11d65fa71cd63f13694a9121491c3e57caa294906d266215f1ecf20bd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1328125, 1.896484375, -0.07421875, -0.01953125], "student_probs": [0.03621963784098625, 0.7491193413734436, 0.10439639538526535, 0.11026457697153091], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "84a1bb4a6e66e7ff8a3f1d205184f6782bb3a6dee808e79df64e2cdfe15fc7ec:action", "state_id": "5a56d331e485a6d92c5937edeb52f950eb42ca59f65bca7b63781a0f3a1ef9f9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.109375, 1.9453125, 0.2578125, 0.4140625], "student_probs": [0.032544855028390884, 0.6904246211051941, 0.12771572172641754, 0.14931480586528778], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2c24b4a74d2e0f1e36d9a97d756896ba6df70c4b10e28ac9567d9a2061b19b1b:action", "state_id": "60e83356669ed3196b3322b908ad7bbe123e66ec299288807259f66e80fa4709", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, 1.90625, 0.38671875, 0.4765625], "student_probs": [0.035136424005031586, 0.6616820693016052, 0.1447855830192566, 0.158395916223526], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c9b034641e296ab727af70213a7e22c297af5c883184199e59b7d98b9eb41f97:action", "state_id": "39e180f2eff429d5b4b9a6d8f7d0f38f1f687c66303d6fa62bd186c809a56d08", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.41796875, 2.068359375, 0.87109375, 1.103515625], "student_probs": [0.04711321368813515, 0.5661627650260925, 0.17099186778068542, 0.21573220193386078], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4fe5208e6d53e779c55d3f427076e766b6ad86d447b4078cda8e392aa465f765:action", "state_id": "2899078341f0fb224bf8e0aa6d820e1ad0b3e2f5251affe745d634f3371338db", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7275390625, -3.28125, -0.140625, -0.177734375], "student_probs": [0.2169603407382965, 0.016877876594662666, 0.3901880085468292, 0.37597373127937317], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "07234ed823ee2de12c3fe109fc7b8582dd5ae1c0390f420b93e5b32ca58ee430:action", "state_id": "4e81656b867d82441cb00516d8cfcfd407fd62d40cb24bf8bf949c2b88a9716d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.935546875, 0.96875, 0.1171875, -0.46484375], "student_probs": [0.08209317177534103, 0.5512297749519348, 0.23523598909378052, 0.13144098222255707], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2c0cf4d97e068feb3a939ee4cecb309c7c226e242d16715ae8a43a725e89bd20:action", "state_id": "555a77fc2ab830345a4a97bd1c1c021605c3a89aa4f701b7d93e696614faa98b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6416015625, 0.4765625, 0.4453125, -0.05078125], "student_probs": [0.11325272172689438, 0.34646639227867126, 0.3358067274093628, 0.20447424054145813], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7cf6db23a879da9c914a07cdb3f7b46c95fd43781d7fdcd09dec30a1fb123e15:action", "state_id": "7489a3f1fcd70551f099c80ae345dd230e6d7aec129a03882f1628cbb277f793", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5537109375, 0.732421875, 0.3984375, -0.015625], "student_probs": [0.11207292973995209, 0.40556561946868896, 0.2904113233089447, 0.19195017218589783], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "11ea55f0cbb156f0dc67da22d49f56a5525667fc1bbadd3c33e0886c7eaef8eb:action", "state_id": "f283aa42b37c7b9101b857bb5b45e20c219c25bf4a76b5c533fbe29c6c1dabfb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6328125, 1.146484375, 0.21875, -0.08984375], "student_probs": [0.09099096804857254, 0.5391840934753418, 0.21321961283683777, 0.15660534799098969], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "625f3225b276becda16f6a3108b81e87dce9f306dc49050aac4ba0ed582ba351:action", "state_id": "d911956830e6dd0e038b510ee1d066c65339a3e5b1fd4cab22317aa93abddd15", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -2.65625, -0.224609375, -0.6541748046875], "student_probs": [0.20301082730293274, 0.04028873145580292, 0.4583863615989685, 0.2983141243457794], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5edcb751b91e20b72eb37d3c9b3a9fc3176bc141813b39fa4b4332caf1a126db:action", "state_id": "29e08539cde96d5da622be9ee00dff421e955b8bdfbf8e9668117a7abcf6529b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.46875, 0.67578125, -2.6640625, -2.91015625], "student_probs": [0.014690273441374302, 0.9267805814743042, 0.032847415655851364, 0.025681717321276665], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "08a1c763911ac2e45c786ff300936d039eacf462b5bf53506f1f1b1da5c0efaa:action", "state_id": "fd2ff14619d778a583ee1d4b00752de5a1e31d056222e9a5ffc0d3d9a44e8845", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.40625, 0.51171875, -2.0, -1.671875], "student_probs": [0.043310634791851044, 0.8014053702354431, 0.0650169625878334, 0.0902671068906784], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "48116dfe74896fc43ac5f0a5cfbd741a01d4b5ca0e3eb5059a4f121921989368:action", "state_id": "676f31eb88315568723a60aded62bba8dbca5abfcdbe238826adb967703e544e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9765625, 1.4375, -1.7734375, -0.8671875], "student_probs": [0.028053531423211098, 0.85250324010849, 0.03437190502882004, 0.08507128059864044], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "de27beb65731abda2899132924b2a72fb267f6349518f75b19f9b97002bc0581:action", "state_id": "44a60628db8d2e6d9b08884169a81096858b51a8c79c895eb8de9a637a7f0258", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.74609375, 1.23828125, -1.953125, -0.8955078125], "student_probs": [0.04179178178310394, 0.8263965249061584, 0.0339764766395092, 0.09783531725406647], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bcbc93f8efb049be8752273aa0664307309466113ad6cf67438f8bcc3f5b2863:action", "state_id": "902f4710c1bf314fecb7640d23d74209395ae7393cfdfafca8b49a6738c2b94d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.48828125, 0.56640625, -3.12109375, -2.05859375], "student_probs": [0.04118203744292259, 0.8736585974693298, 0.021871615201234818, 0.06328761577606201], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f97b1ab1e539000058416641e8a1f5bcfee9fbfaea93f4fbc859a89f4f832b32:action", "state_id": "ae1b4178bb7a951370cd1215c64a913fd970cb6142d005efc468237ca98c4b75", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.7109375, 0.61328125, -3.12890625, -2.2890625], "student_probs": [0.0322992317378521, 0.8971850872039795, 0.021265259012579918, 0.04925044625997543], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "13fa930d1ae958f1ccabc1ef0cafe88f7285b2307248d9e3217d10d35188c73d:action", "state_id": "acf37ec0320032654b736b0701564ab8412259b0097f51f2ed3c0a4763f2acda", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.67578125, 0.56640625, -2.74609375, -2.1015625], "student_probs": [0.03413262218236923, 0.8734414577484131, 0.03181510418653488, 0.060610756278038025], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aa2ca3b8da898ccba64495abe3262dd5eab68c2893ed1c2865528b3b84c592f9:action", "state_id": "5d78591c34d21ad6fb6ff4c899a4290a8bfe3ebb4c439c6e8257ff9f4d90bfd4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.5703125, 1.171875, -2.671875, -2.02734375], "student_probs": [0.021827051416039467, 0.9208871722221375, 0.019719095900654793, 0.03756672888994217], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f7ed5188e34f9e456d1b8803392ac193c6a6a24c72c0df61ac5497e93d3bfa2b:action", "state_id": "e24fbc5f48cc1d3f39504e4d156c99213576960d129217b09c473c134f16c9dd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.375, 1.615234375, -1.51953125, -0.9267578125], "student_probs": [0.04287920147180557, 0.8528820276260376, 0.037108857184648514, 0.06712986528873444], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c1729fd88465e545bc23a1cdc4c23cf07504797598376d8bdec9c9dfc14bdaee:action", "state_id": "6498b18728b8d8f40a7a54f3c683282a43e290ea22990eadc3620b73a276905b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.51953125, 1.806640625, -1.43359375, -1.107421875], "student_probs": [0.031815383583307266, 0.8854728937149048, 0.03467043861746788, 0.048041217029094696], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bbb69167f29a27e61f78124a6b2d647e2d40e97f73505167ae747cab0250a8a0:action", "state_id": "14d26a472e227e3441871bb6f8173b6d33ac4897320da62b0f6cdd3c72a5b390", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.83203125, 1.765625, -1.53125, -1.1796875], "student_probs": [0.02451971359550953, 0.8952775001525879, 0.03312402218580246, 0.04707872495055199], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c3ef01ffc787a195b1f22fb507aa7be527620b7f7b6fe2906f6815c39870dddd:action", "state_id": "df3da8f29b5f6987585a48b70aea33d017997ad0d5cf8e57c01690932704bfd3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.875, 0.765625, -2.13671875, -1.60546875], "student_probs": [0.05847596377134323, 0.8199478983879089, 0.045010555535554886, 0.07656553387641907], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a70389bf37a028470095778ca8ec1d1b6ea00208d034c8f767104cf02587557b:action", "state_id": "6038809a99377a5953f90c4022954ccc2fceca81f54ed35c3155352a27c294d5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5859375, 1.21484375, -2.06640625, -1.3984375], "student_probs": [0.051861245185136795, 0.8535063862800598, 0.03207583725452423, 0.06255660206079483], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1c188a037c39fadaefc1546eb26763a11457108def72455df7b8eaebdca79271:action", "state_id": "95a9f7bc8535ad83779d3ce87ef56aa65e267278baadfd7c5b62949ea39b4396", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.41015625, -3.03125, -3.0546875, -2.0234375], "student_probs": [0.2829328775405884, 0.15203578770160675, 0.14851388335227966, 0.4165174961090088], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b34eb743c8ea8c3e09bb5ebc119400ac03647d717e5bacfa4fc720321d99af81:action", "state_id": "6839c61fa3e3d1193eacd41b0769f0e2698d874627f31f26766363085b0de219", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4140625, 1.1953125, 0.25, -0.474609375], "student_probs": [0.044583436101675034, 0.6059135794639587, 0.23543263971805573, 0.11407037079334259], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e3ae8fe9b14a55a9bfd3a717fae4dc54cd054d116efe3909cb6b2c2830cb8665:action", "state_id": "986aca197c381178da1cbc1fa76f1ae2c9701a63880ea91c0516ddb1a0dcb0bd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.587890625, 0.6484375, 0.5390625, 0.4296875], "student_probs": [0.09712817519903183, 0.3344072699546814, 0.29976072907447815, 0.26870378851890564], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "80aa511b0a5e08d53fa5333a6ab764c2afbb4e821f0912b163ca828ec0bbd4c2:action", "state_id": "ee4a814758812cc6a521ee5366b49484fc1252a7b57c0c0c64f998fed2f733a7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.080078125, -3.62890625, -0.328125, -0.529296875], "student_probs": [0.2026786357164383, 0.015844041481614113, 0.42990949749946594, 0.3515678942203522], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "47a7489c6bc9840ba648f68fcf05babaee79be1505a8673f853bf5e14d1c6c56:action", "state_id": "bed8c004a1503503026f99ba93d30e928f8f5ab19fde1807a5bfcc59661e9765", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.734375, 1.189453125, 0.00390625, -0.529296875], "student_probs": [0.03491988405585289, 0.649942934513092, 0.19860891997814178, 0.11652834713459015], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3a4cc8cfa21b2c5533795119090248a494f6250e59fc3684d270bf293b99dc20:action", "state_id": "db632059096dd549c94875dddc6997ec45f50ec6fdf3b5f93a257a927d8ae5f0", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8916015625, 0.642578125, 0.39453125, 0.44921875], "student_probs": [0.07646159082651138, 0.35459214448928833, 0.2766965329647064, 0.29224979877471924], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "be94fe33831fbc9ca57b4bfa12efb3c9ecc3fbdd00747d85ecaaa57fbba03f38:action", "state_id": "0c4915fd3ba296a11f84d2687ed998693050a97c3575c7e156eb0ee1ca75631c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.79443359375, 1.4375, 0.40234375, 0.546875], "student_probs": [0.05730217695236206, 0.5339339971542358, 0.18963780999183655, 0.21912607550621033], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "15141f75e26da45a316d65137e0d996408cde8e3cdf29a187640296e9755e89b:action", "state_id": "a4d1de5f7f90691ca9bc74b5693c7e99f7c73ba0d866b1e216789d693f6931e9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.15234375, -3.3515625, -0.3740234375, -0.50732421875], "student_probs": [0.192502960562706, 0.02134660817682743, 0.41923511028289795, 0.36691534519195557], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3ea738cd505988b4115a138b36b9ab754aef49861f5d0de9230c4e42a2d671f1:action", "state_id": "fd0cb56cc4c9e37df4e207cf5ba81d1f8535a51bd042a45d45509b54e4b214e7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.392578125, -0.861328125, 0.751953125, -0.388671875], "student_probs": [0.1732902079820633, 0.10844224691390991, 0.5442991256713867, 0.17396844923496246], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d6d7ae9afefcfb27202d2596c123bae5e499ffc8ede491e75d7b1c799ef3b76c:action", "state_id": "3943d2877fe24cf60aecc7e5151585a37fd73a5b5ae8c31631f55b46d75b1790", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.970703125, -0.01171875, 1.27734375, 1.267578125], "student_probs": [0.24516397714614868, 0.09179018437862396, 0.3331416845321655, 0.329904168844223], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ddfc03dae819d18f60e4ca094a8eed78e9835df66f083e47ee63bbd8d0c6bb05:action", "state_id": "8ae201bafe2ad90be61161f382f572f2accd8d3930e8eb5eab2c5dc88501379d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.12890625, 1.3828125, 0.94921875, 1.046875], "student_probs": [0.10776545107364655, 0.37761056423187256, 0.24475793540477753, 0.26986610889434814], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3752decfef2668bc7e0c3c196a75873623b50ebc3c45bf00b5a4bfa2e1fe60a5:action", "state_id": "efd286ca36f65d0274c15ad7e0253a67ba81698e7675e44272237a2ae447245f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.181640625, -2.26953125, 0.3671875, 0.171875], "student_probs": [0.23368652164936066, 0.028965050354599953, 0.4045634865760803, 0.33278486132621765], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2d7725192c9d931efcda8a69e895a962699794cbe4e08db7bea69bfede036de6:action", "state_id": "f4c0decda3dcb7e56581d0ad7af52085e23c0b513c50200ee68b80cb23e9c638", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.4453125, -0.6075439453125, -3.29296875, -3.1015625], "student_probs": [0.04842051863670349, 0.8269069790840149, 0.056388624012470245, 0.06828387826681137], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3667e9d0f5ac225cb3a49f8a3f8662143e38a6a2b66d75cf4a7c7df232984366:action", "state_id": "e0bbd2e182f2080ae95bda4f82237543da7fbbdd07dddd61e0247521d997d15f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.984375, 0.29296875, -2.58984375, -1.47265625], "student_probs": [0.07713240385055542, 0.7520984411239624, 0.042100295424461365, 0.128668874502182], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "351e209ead4f521f7e647d2675044b3fa94610892963001e03390401346222ae:action", "state_id": "59852d25fe20b82ddf31565a9a514238adcd1c9def667e43ba6d134f4982a660", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2421875, 0.38671875, -2.69140625, -1.67578125], "student_probs": [0.05794193968176842, 0.802994430065155, 0.036974288523197174, 0.1020892933011055], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9b0106ae5a60775baf4845ec031a542865a0a18cea4405794a75bbac67d52fd6:action", "state_id": "e41dc0aa6bb21fe0ae02b69dd92b2e07d597f191812f6b3493c64cee2a965069", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.24609375, 0.6171875, -2.6328125, -1.62109375], "student_probs": [0.04746885597705841, 0.831602931022644, 0.03224474564194679, 0.08868349343538284], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "79d477a4fcd44938e66778adbabae10857b69143e88969db115b74d3abe67b01:action", "state_id": "46e9b7abd6db32e537267ecad21d9c554542762973cacda74e9adaaaf8d2b0d3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3046875, 1.2578125, -2.35546875, -1.69921875], "student_probs": [0.025618817657232285, 0.9030944108963013, 0.02435034140944481, 0.046936508268117905], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f95a1109736eba8eaa96296bf54c8b5bfc16a5c04873237c44d03e533565e9bf:action", "state_id": "e726f446d2c518e58bab1ae57771275dbb980e7721b6b1289a540cec4b237c0e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0625, 1.765625, -2.15625, -1.15234375], "student_probs": [0.01985250972211361, 0.9127439260482788, 0.018075913190841675, 0.04932774230837822], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bd31724eb427c14d6a43a7cfa3a2010f2f957647af2f9a42803e8d6af221d54e:action", "state_id": "820564f3a0c422014944b7d6868ee68e44cbeedc5155f6954ab855bbe617e75e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.73828125, 1.806640625, -1.94921875, -0.87109375], "student_probs": [0.0257552657276392, 0.8920845985412598, 0.020857250317931175, 0.06130287051200867], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "49491081e76ab19432163ffd637a6c0f3515e4f84949d3c6fe13aee1306497e7:action", "state_id": "009e9bf27607357c8bebb34e60384db38283c09aa3e242d6f5f801ebf49eb063", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7734375, 2.001953125, -1.31640625, -0.4853515625], "student_probs": [0.02007235772907734, 0.8754466772079468, 0.031701844185590744, 0.07277914881706238], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9f90054689a6f395db99038b537214ba031145e0175ec0f34392acae74d2310e:action", "state_id": "21c4c8ea558ffe7494480ec34096118fd49345c9ac5004ec79075f542b4fb5dd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.36328125, 1.69921875, -1.208984375, -0.77587890625], "student_probs": [0.03945226967334747, 0.8435266613960266, 0.046034373342990875, 0.0709867924451828], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "27033f35b7023c008fe9d8e7e251003630ccf10dfdcce3368f71ca93824edcf9:action", "state_id": "7a1b2873d6648c98f045e37cfa84e9b9275d9257df29d32f8b4a2fc825533695", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.80078125, 1.666015625, -1.59765625, -1.44921875], "student_probs": [0.028026524931192398, 0.897800862789154, 0.0343388170003891, 0.03983372449874878], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "95c32c296d5b798a265883e8dc8c171ec80366d0d7e16528016fbc269f1250dd:action", "state_id": "1ffac8ce95e68e82d38215bd1ad7c4564501d99d45b2b5ee0ce03d8706988150", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8984375, 1.791015625, -1.7734375, -1.44921875], "student_probs": [0.02287115715444088, 0.915371298789978, 0.025916418060660362, 0.03584110364317894], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "57b655e9855bea7bf5e689e85100bc3edb39f0614a8fe46b21625d9a3921c2e4:action", "state_id": "d3c72aeda92c807c3cf9775e57c1f850c8280b0a7ede11034bb90368fa7c59ca", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2265625, 1.3828125, -1.828125, -1.60546875], "student_probs": [0.02421693690121174, 0.8946452140808105, 0.03607102110981941, 0.04506680741906166], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d77e055632aab0ef0b48a57227a316b33387d02453379875572437c06e48976e:action", "state_id": "fb348e21a55ea804033f29c84bf0e1c5c6b23721e90685e7f0af43b713f2ab23", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.20703125, 1.43359375, -2.25, -1.5703125], "student_probs": [0.023830026388168335, 0.9082970023155212, 0.022827766835689545, 0.045045141130685806], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ff36f853ca1cb41f1836a3fbfe65ab88c371b2ba76432960943e4d42b5356b64:action", "state_id": "d8dc121af61655dcfa47ce69dd6a2a517ae349b3174cc720e72f5416ad6eaa18", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.4609375, 1.375, -2.18359375, -1.89453125], "student_probs": [0.019834032282233238, 0.9190465211868286, 0.026173384860157967, 0.034946050494909286], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7b3a76ecd4ca33dc0d44250377d86ed43e5ab56c8e43060d9c65e3196259f9e1:action", "state_id": "51d63321c5060080b827d85c31c19f3ae932f980dfc03a7090216b59a275ebe9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.078125, 0.82421875, -2.2421875, -1.60546875], "student_probs": [0.04614732041954994, 0.8406561017036438, 0.039164721965789795, 0.07403182983398438], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e3d1a7aed554a69205ae1c9e8a9a706aaddd9a3e3573bad26d33af9908e3990e:action", "state_id": "5c4209f76de917bf30b484aa00372152788538e2fede540977b007eebdd3c1ea", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.203125, 0.80859375, -2.2890625, -1.59765625], "student_probs": [0.0415419340133667, 0.8442276120185852, 0.038121022284030914, 0.07610943168401718], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ef531109712647a305266f0842d8c3a5bb31174986dec67701f0ddbe3f6c5e81:action", "state_id": "c2f332121fbe0ef4c2ac2c4601edab502672ed932856a62c46719a5949045349", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.53125, -3.078125, -2.921875, -2.3125], "student_probs": [0.28572165966033936, 0.16536301374435425, 0.19332894682884216, 0.35558634996414185], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b3b8fade132c310a0d033c1ce33476d818de13559177637be5c034653776c22c:action", "state_id": "d366341d86a8bebe0ab89f6ff6bf55544401748a3e12ddd71c62593cfc5cf6cc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.421875, -0.55810546875, -3.359375, -3.0625], "student_probs": [0.04756378009915352, 0.8336728811264038, 0.05063138157129288, 0.06813196837902069], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d1d9e09e1c2d1fff59bfb6bf33781b826e1c5cbadb7201bd96b6f85d0fac49d4:action", "state_id": "df79842f4cf4fd5038d1966e381e020b2db3fef2aacf5782d90b72e190b9374b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.234375, -0.5439453125, -1.169921875, -0.212890625], "student_probs": [0.14623171091079712, 0.29166969656944275, 0.15596716105937958, 0.406131386756897], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cc57ef19363258f6ddd39363e4985f67c3070a93456bb2b2f31b2271489d491c:action", "state_id": "0ea57336a7ae157b8052f6eee4edf87d78b53a7ecb42f4147b28daa92a1aa7a9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.216796875, -0.11328125, -0.776123046875, -0.12890625], "student_probs": [0.11714393645524979, 0.35315921902656555, 0.18201282620429993, 0.34768399596214294], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "311da6306cda9e6d65c5f024134f66861555d0d5c3e8770438b692e2f1489a9f:action", "state_id": "a947fde26a98dfab678d6d259bc82e6414f838360f314a100a801eb87e473668", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1484375, -2.30859375, -0.8154296875, -0.55859375], "student_probs": [0.22161637246608734, 0.06946281343698502, 0.30918988585472107, 0.3997308909893036], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "65cb7ab8a0ed1c0e7fe058ebd92321c08b16baf27bcb9f90b604ab9415574816:action", "state_id": "9def6d2ef3ef3eb39d72c0ba3b80c634550d36ee22721ef031e148538ece795b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.1875, 0.25390625, -3.671875, -2.84765625], "student_probs": [0.029195668175816536, 0.9118053913116455, 0.017986929044127464, 0.04101197421550751], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ad3fbe0d54d381be40d46994d13d64284a144faeedf9b39bc68ad4a33ef9fd1d:action", "state_id": "7ec339c8542942a940bcc37411b1efe560029f24a3775c67d6564ebf3d0ac75d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.21875, 0.1484375, -2.6875, -1.83203125], "student_probs": [0.07264657318592072, 0.7749462723731995, 0.04546106234192848, 0.10694609582424164], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b5c85f46a54e0f313fd6b7d0a6c408b1278e15ca00ac96fff0f73c0a08cbef77:action", "state_id": "77ae8f3672c139705d75d937e6ffe55ecb0028023115b3a0dd8619cf3a9c95b4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0859375, 0.06640625, -2.66796875, -1.546875], "student_probs": [0.08418809622526169, 0.7244387865066528, 0.04704112932085991, 0.14433197677135468], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "56c6aa589b71cb6abdae67847580a774f98b24981ec8dc54251ab105f1a6c4b2:action", "state_id": "33e3f6988ff0296bb884e0b0c2bbcfc71e7f615526ade61c26622f133009a4ea", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.93359375, 0.796875, -2.56640625, -1.328125], "student_probs": [0.0534665547311306, 0.8201809525489807, 0.028395870700478554, 0.09795664995908737], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0f13b646d08b76b68b4a5874e1caf83ed543e79821da9be4bccbd086054e8d3d:action", "state_id": "882035106b17e83a5672d6ceedb2737ecdc069db53f2789c26f3d24778a40931", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3828125, 0.79296875, -2.83984375, -1.90625], "student_probs": [0.03677929565310478, 0.8806993961334229, 0.023287199437618256, 0.05923411250114441], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a8913e5ca7e8c7d5479da2efdb202abe5631ddbe1329124dc74716656399ee81:action", "state_id": "c7548e7203f53774d1b15b9c2f1d6b2931fd60f0ec28f2a144f9f2421d98f572", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.31640625, 1.81640625, -2.26953125, -1.67578125], "student_probs": [0.0150832524523139, 0.9404866695404053, 0.01580711267888546, 0.028622983023524284], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2cbe801fa278d025ad343b7c3b29cfa1b78afcbca497c6b6c9d2d0eef92e2093:action", "state_id": "f3cc69a22ad827137cc91ca8c7c29bc508e0673a6fc8912bf0ab664a0164d014", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.38671875, 1.66796875, -2.3046875, -1.90625], "student_probs": [0.016294749453663826, 0.9396715760231018, 0.017687782645225525, 0.02634587325155735], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0277179d33c01592cfb528129d0953227041ff004ff2cbc2693c1e403cfe94b0:action", "state_id": "81f98ca58f0f74a48e14ba2e5440b8f4db9ecafa31003562306592a61c8520e1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.41796875, 1.83203125, -1.9609375, -1.8515625], "student_probs": [0.013432425446808338, 0.9416857361793518, 0.021214880049228668, 0.023666908964514732], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "59a8b1e36482842240d17be70ef3b856f239316cd8e93320cb1eaf3b020e11f7:action", "state_id": "dfaa7d39862aaade22d55e57bfacd62534c9af6ebfcddb1dbe2abdf9e133adcb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.71875, 1.6796875, -1.83984375, -1.154296875], "student_probs": [0.029795823618769646, 0.8914110660552979, 0.02639763429760933, 0.0523955300450325], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "37b57cdc71c68a943f8637f3ff17bac886a32e0dce50065dd30dd288e474d7eb:action", "state_id": "5f4cfef3ea0beb902aa612427ea65a84dc6951916ec53d19cb7ad3fab4f23494", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.75, 1.44140625, -2.25390625, -1.21484375], "student_probs": [0.036186669021844864, 0.8801541328430176, 0.021862756460905075, 0.06179652363061905], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "33dba36cdeddf8f6cba8c702d949ec7043f7f1e96c48daa01bb94f7f29cac814:action", "state_id": "50ef3991cae276a0eccb1f72fd1f6328a1ae6368e84c70b52359f71071f485fd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.24609375, 1.39453125, -2.31640625, -1.76171875], "student_probs": [0.023997554555535316, 0.9146823883056641, 0.022368179634213448, 0.038951873779296875], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c3a3a415dff965d13c7a0f273fc00eba90302838d6e281e5e9a58b718ea37b9f:action", "state_id": "258f8bbdda97be3bc8b3bb0703863d1498ffa3d891e5e088ede7c56f936f652f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.75, 0.7109375, -2.15625, -1.57421875], "student_probs": [0.06861481815576553, 0.8038768172264099, 0.045707326382398605, 0.08180102705955505], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7973154c29ac2bf5493759a7bab1df227f2e2c97da5851480eec7005ce29e509:action", "state_id": "e02a7d5b84f397f717ed2cb6ab9b4e719d7746a34614217769e004d1d5654e44", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0859375, 1.10546875, -2.3984375, -1.81640625], "student_probs": [0.03654493764042854, 0.8888681530952454, 0.026736848056316376, 0.04785013571381569], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "69464a4253f651ba5f2b9611519193ab03c5ef48c0be645c117c85bc3f59cb35:action", "state_id": "8d9423f1cd3514b01d093a864556607474e4e569e5c661c846e99d4863b0811c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.77734375, 0.73828125, -2.30859375, -1.3203125], "student_probs": [0.06434348970651627, 0.7962082624435425, 0.03782558813691139, 0.10162271559238434], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c7357273cd80061c9d597308ed1f9ce1f7b61bc1da0cd9dc9421a950745b4798:action", "state_id": "cd520ab50fc2ee3dd87dc26e8f3d4384048811372785f94aea21838626feb38d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.23046875, 0.60546875, -2.6171875, -1.83984375], "student_probs": [0.049496229737997055, 0.8437312245368958, 0.033621903508901596, 0.07315068691968918], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8706083f039091256affa21d624c68e7a2804d7f5393ef951045988dc58aa10e:action", "state_id": "83d43fe98942a96b13ad75efdeb3433ecd1f9cea54684c713d62c70413c188fe", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.08984375, -3.28125, -2.8671875, -1.88671875], "student_probs": [0.3345998227596283, 0.10164933651685715, 0.15379053354263306, 0.4099602699279785], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "118930552e9c139348feebe5965ab3525ea5c89154a012335157e990b055aad3:action", "state_id": "714b687a391194fc2a3d05346a4b437273420fd76bc7de0592b08eab06ec4c98", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.88671875, -1.57421875, 0.17578125, -0.83837890625], "student_probs": [0.1836225390434265, 0.09233120083808899, 0.5313293933868408, 0.1927168369293213], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "83a48470fa07c40c2890d1244b646764d0a8bed51449d38b17c6a87a1cce40de:action", "state_id": "20fb4207019ecfd94d1e6f86acff0367e3a7690a1df80a5b5722c0d9156b45ff", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.02734375, 0.32421875, -1.890625, -1.1328125], "student_probs": [0.06624859571456909, 0.6957404613494873, 0.0759543851017952, 0.162056565284729], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a31dd64c0668b86c40f24bef47c20aed79010203d68188d73507f05f96489d52:action", "state_id": "4cb42f1fcedc9967cd13d42a3da0f3aa7e4b60dff1ee2b5c72d6341203600639", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.12890625, 0.296875, -2.65234375, -1.3046875], "student_probs": [0.06586034595966339, 0.744950532913208, 0.03902096673846245, 0.15016809105873108], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8fee97d7427cb4b82b4cc6b8570ac219552468358f4103e6bc1be6d70858cc4e:action", "state_id": "8d5622d7c26325251d4ac734708c2488bb6b4ac0f77a643adb522a7f92b355cc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7421875, 0.38671875, -2.4453125, -1.111328125], "student_probs": [0.08489015698432922, 0.7135584950447083, 0.0420236773788929, 0.15952768921852112], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6acd678e6ec4ef5ccf9a8f8de6478377a59eb419f098e6b90e4aaa46728e6730:action", "state_id": "e46c5d6feecbfb67b717c01078176ac95cb57c4f5a06a8d3e34431d49da33703", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7265625, -2.62890625, -2.71875, -1.71875], "student_probs": [0.35916629433631897, 0.14568427205085754, 0.13316620886325836, 0.3619832694530487], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e02e37dd00ef87655ae0eb326983cb891960e3b8121dff1719a00072e0a5fd4e:action", "state_id": "0ba63b84d3f417e3cec34c87df4533212619243a32f240bf144c2c1a2fd5e42d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.61328125, -0.15234375, -3.1953125, -2.80859375], "student_probs": [0.070936419069767, 0.8310762047767639, 0.03963658958673477, 0.058350712060928345], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bb41a05d96810551e9d9a66041cfb5890d7812597056cfc22b9cda3ac73bb158:action", "state_id": "9a849f804590d04f1d11f576f805ad0808f82756431d4196d0064a83067e6ea7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.984375, -0.390625, -1.1328125, -0.419921875], "student_probs": [0.18411779403686523, 0.33339422941207886, 0.15871945023536682, 0.3237684965133667], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "92e1371aeb667765fb33bd46703fdbf50d829955bb19e9887aabf9b359cea7fc:action", "state_id": "642045ea4ad6cf304172ee59d037d7c1f0c0e82d6b3dc99eda180d434d639536", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9521484375, 0.140625, -0.283203125, 0.15234375], "student_probs": [0.1117018461227417, 0.3331546187400818, 0.21806180477142334, 0.3370817303657532], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2fe3dde724d3729f017ea708bca981c144523a407c385195e96f87304152683d:action", "state_id": "e36e04cedfa3a15a954db64f78d9bae4ffb3deae8ac866f15aef850959b8178e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.19140625, 0.30859375, -0.1640625, 0.47265625], "student_probs": [0.24096617102622986, 0.2709255516529083, 0.168879896402359, 0.31922832131385803], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "29f0a3008e77c169cb1cfeb497893606fb42fa156161beb5601b95125d8eff89:action", "state_id": "b55bcbc5a2602e5aec4594819cdba31aba2fb02e0a845f7bef9b9f172e49a1a9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.37109375, 0.669921875, 1.021484375, 0.83984375], "student_probs": [0.1705738753080368, 0.22998099029064178, 0.3268688917160034, 0.27257630228996277], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "15d750c1048e0060ad689a47d6067a2870546068ba09b7595100946b18f3ef64:action", "state_id": "fd70b248dc6e76507cebcda77a6af79cb80e62209f9b5d67115099d1cd298adc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.2890625, -2.7421875, 0.42578125, 0.31640625], "student_probs": [0.31031981110572815, 0.014974568970501423, 0.3557834029197693, 0.31892216205596924], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a799b644c90d86eff63cea7f1298e2db3c12cf40c9750be01e6c1745871a9be7:action", "state_id": "6c3bd467b8d55fb6e708c84d070fbf26966676a100abdbb0fdaf233e975a561a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.61328125, -0.56787109375, 0.40234375, -0.78515625], "student_probs": [0.07331913709640503, 0.20856104791164398, 0.5502906441688538, 0.16782917082309723], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5a77912cfe5541d32a3f118f5b7b861c7fa40631419c89edf6ecd45df956bdb0:action", "state_id": "acb46c15bbf6eb5ebaef849e29144ffc6e13b30ffee3f9d06e7874c40545293d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.009765625, 0.2578125, 1.029296875, 1.23046875], "student_probs": [0.26751405000686646, 0.12611812353134155, 0.2727902829647064, 0.3335774838924408], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "772387c0c830b7c49cc7d67d809507109c0e902070fb719820c0118f8ceaf0e2:action", "state_id": "efe2d0916170ec444f7ca9bc3f1453eb1dfc8d2d4547e37d05a0abf45b69ad30", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.59765625, 0.921875, -2.1484375, -0.3994140625], "student_probs": [0.05775820463895798, 0.7175170183181763, 0.03329756110906601, 0.19142720103263855], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e0cd6a7a4fa46be708f89a28118e74df0450007da4712af65b42353523045e5c:action", "state_id": "d996e8511c1c575077bfe9bc77a2b33658bd621f96627b3025c18d9886125277", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.59765625, 1.015625, -1.7890625, -0.76806640625], "student_probs": [0.056300219148397446, 0.7681458592414856, 0.046492550522089005, 0.12906140089035034], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e15928e5a6d916a9639f426583d9e3b292e18f363b53a8e4249b920f7e367003:action", "state_id": "5711341dd476a7ed909365566454a88ddc3f2268b6b8019351c447d8673d7ecc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5078125, -3.3203125, -1.30859375, -1.1953125], "student_probs": [0.26662880182266235, 0.04352595657110214, 0.3254068195819855, 0.36443835496902466], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c9e84766c7254b60ce861286c8d5f890d738e86b2142106874628d71e5371769:action", "state_id": "61c2c1ecbde8539560f4010f1c3213d5c48b0eab9fde403dded65b627d978df9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.0546875, -0.6473388671875, 1.01171875, 0.27734375], "student_probs": [0.1708941012620926, 0.09448044002056122, 0.49643391370773315, 0.23819158971309662], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "535e2cfad59a691664deeb05f5304ac0eedc59439022e40ab6753450f15dd6f1:action", "state_id": "0094f2d966c3c9a502bdd429ad7aa5bbc74a9b9ae8cd2bed9c76f71615582bb8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.1015625, 0.2890625, 1.75, 1.552734375], "student_probs": [0.2029859870672226, 0.09007447957992554, 0.3882208466529846, 0.31871867179870605], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "af0a25d1ee8be6b7833889bc96c1a6b9e6a5db33bfdfaa47f5b75556ef4313b5:action", "state_id": "485c440ceea13dbc4ccb1c3bbd8f3ab2768d8bf5dd16aa9c520804e5f5811082", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.44140625, 1.4140625, 1.275390625, 1.34375], "student_probs": [0.11886634677648544, 0.31439682841300964, 0.2736867070198059, 0.293050080537796], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "42ae7c07a0cd01992f968ff12fc5a3b7b44ec8d607dc552af278433ab2b4e3b5:action", "state_id": "8ee0e75341ceae8fc3e61abba3c03df6203abcade5101031ab99acc822ed0629", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.33984375, -2.76953125, -0.0234375, -0.05078125], "student_probs": [0.2634749710559845, 0.023202748969197273, 0.36153706908226013, 0.35178521275520325], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8893be5a836f7cd0bdc640557a2fac49d58c91b1a86d971ac4b8c54a582f8d70:action", "state_id": "4430083538df94ba5700736ca36431d44794d4b9356f4cece40ec91fd2db643b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.193359375, -0.9296875, 0.77734375, -0.154296875], "student_probs": [0.19385464489459991, 0.09283098578453064, 0.5117374658584595, 0.20157693326473236], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "19e7be716a5da888237280e4768c365ab1a23a683f6d9512326ab5215bb7b54d:action", "state_id": "977044442b1f5d498f965bcdac6945e2494ece47d8c134782acdb0630d5275c5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.32421875, 0.33203125, -1.123046875, -0.236328125], "student_probs": [0.09587299078702927, 0.5023385286331177, 0.11723683774471283, 0.2845516800880432], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1ea722a5c279bde43f0432a86080889af14a6df4d6797ad7fd24820a45923f62:action", "state_id": "ea9ba2cc29d6afa829cc10f5f7ecc9acc0714b4071d46e82a425ea4d96135a14", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.93359375, 0.20703125, -2.3203125, -1.34375], "student_probs": [0.08341855555772781, 0.7094540596008301, 0.05666473135352135, 0.15046259760856628], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "42b6500229753cf96550e6d5bafe67ea9c6722ef5cbeeb6e7409ba4b4bcaf544:action", "state_id": "92ff40e667e2b8a78d3afe42fe373005db6a44bdaa9d6c75c1dc8759d8536a49", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.734375, 0.39453125, -2.3515625, -1.24609375], "student_probs": [0.08639577031135559, 0.7262141704559326, 0.04660702124238014, 0.14078304171562195], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bde76b7f50c4333366bf9ecea9bbaf9c0e60aa49fb6250f3b8c934f22fec85cb:action", "state_id": "fd8feb6454f637143c169ab8caa6b9043a762dc817bb7bc859387dc99c6c3f6d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8203125, -2.8203125, -2.7890625, -1.82421875], "student_probs": [0.36449286341667175, 0.13408944010734558, 0.1383458822965622, 0.36307185888290405], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1d363089380a0da07a21f6dee8652920a4a6845c7358e3c49bfedc1c8b79b301:action", "state_id": "5cc043fb5705f7b29400952fb8d5c0bb3ec85708b98bf69f53c15535a9cc9a63", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3671875, 1.212890625, 0.328125, -0.4619140625], "student_probs": [0.045209743082523346, 0.5966858267784119, 0.24631842970848083, 0.11178596317768097], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f3019dc7ba1472919d358a295284fd6af9bfe02c1bb5315b667a6b71c39db4fb:action", "state_id": "6c846c7e81a28f33424a1a0fc7f97e980221d78689ea64c49d382ba973cedf72", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.509765625, 0.6328125, 0.64453125, 0.44921875], "student_probs": [0.10085038840770721, 0.31614986062049866, 0.3198765218257904, 0.2631232738494873], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e86ebb08df4236e70b9fbdd0a4d9b012c5c1206d11775925f059ad65cf2253b7:action", "state_id": "dd83cbfd8d53810fe762ccc518941d187bf8e872cecb755c507eb51f5b371d41", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.11328125, 1.298828125, 0.00390625, 0.37890625], "student_probs": [0.05086332932114601, 0.5675061345100403, 0.15545085072517395, 0.22617967426776886], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7db1e2d1a6ba479dd843d6004c4f71d1bfe28b1b2e8f664ca5d6fb4a202ea6b1:action", "state_id": "210f757705905dd36c3dc57712ea43d55be0bb5d2a268c1dba8d9c8cd5f99e89", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3046875, 1.732421875, -0.314453125, -0.01953125], "student_probs": [0.035521455109119415, 0.7404412031173706, 0.09561897069215775, 0.12841832637786865], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "07060adb464e504dba6417c53a1fd049d81dac1b746f69e44319e2bf585eb946:action", "state_id": "8daa521c4ecf81b9cd0e8ad099ab17aba10ee0dfcad99e82295483e7b0959cd1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.640625, 1.853515625, -0.931640625, -0.6842041015625], "student_probs": [0.02593611367046833, 0.853868305683136, 0.05270027741789818, 0.06749524921178818], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0e27770a4becbe1f37ff0dbee733938ad62c8632aabb7054c7986923ddc013ca:action", "state_id": "9645253bdaa73738b2f86399f73e4bcefcc22548d54cc28b4bd42e8b99421bd3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4765625, 2.1171875, -0.53857421875, -0.103515625], "student_probs": [0.022793371230363846, 0.8289996981620789, 0.058233343064785004, 0.08997363597154617], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "919977152b6908ea2a44b975e43b43fc60909ddc6a83fd5cf35dbefc693333fe:action", "state_id": "b146ead82c4a8cee49bc72e5cabcfdabebc3bd1bee2e0dd9a8bc0333e8d09de1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4140625, 1.8828125, -0.515625, -0.099609375], "student_probs": [0.02923419140279293, 0.7901430130004883, 0.07179224491119385, 0.1088305413722992], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "47fb2fe6fcb977e70a79a4571281570f392290abd20c7d518f7c2f7f0a21ac59:action", "state_id": "1c790bf3e95dabe09eb6d3963eb3da9dd3e56b5e6c2d5d4370f59fd5fcf0bfcd", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1796875, 2.0, -0.5263671875, -0.0390625], "student_probs": [0.033233772963285446, 0.7989146113395691, 0.06387236714363098, 0.10397927463054657], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "37b0821e48486f71f18415313bd5ae83c65af01326c5612e53acedcd04fc28fc:action", "state_id": "e1c04a7577b2ecf2ed8a33d605160f0bdcef083c079ef9f1cd31e03a8975eff7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.150390625, 1.69921875, -0.80078125, -0.44140625], "student_probs": [0.04601621255278587, 0.7952075600624084, 0.06527461111545563, 0.09350156038999557], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c5c701ba0188bd55c88ac218c2992c68fa57ce8632dd6403e0836f9a272a2c20:action", "state_id": "2b4009a82ba3dc881a7d22d8376f34a46d032b544dbfe6988f018d8144fdbdfa", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.046875, 1.7890625, -0.611328125, -0.296875], "student_probs": [0.04606345668435097, 0.785214900970459, 0.0712052658200264, 0.09751633554697037], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "db4ff4b59e355356c0492c9e3608f210e53790fc3edc652b70bf924ab5edea35:action", "state_id": "077271d7171f1d526351cfab9ef7c5d8b5d47c46a3caa86368fe16e07a170c10", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9306640625, 1.98046875, -0.4912109375, -0.078125], "student_probs": [0.042964447289705276, 0.789583683013916, 0.06667473167181015, 0.10077719390392303], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "85047e1391c06f1bb7ae60a6453d94841163267e73fad4640435b61037938e2a:action", "state_id": "690edc5489a7c0bca6d0035553b009a61bf3402a1dd8baa16f8198927a967c2b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9228515625, 1.75390625, -0.31640625, -0.03125], "student_probs": [0.05047747865319252, 0.7338356375694275, 0.09257069230079651, 0.12311622500419617], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f290625138aad85e746a5b1f3e2332527d92db84854461f05577871099cd14b8:action", "state_id": "480fc18254aa6acad24243859bf69c001d1fff78842098bcd9f6910ca4698bd9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.66650390625, 1.021484375, 0.42578125, 0.50390625], "student_probs": [0.07928339391946793, 0.42881128191947937, 0.23634999990463257, 0.25555530190467834], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "db0a6bd638170fcfe0cb0a2ad60fcfba4dcf186f1a02d919252c525dde732cbd:action", "state_id": "b7e74c4537c4ee6b8d9c09d412f4c4c1349d1b6c900cb2d510a895977652db49", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.19921875, 0.892578125, 1.4375, 1.3359375], "student_probs": [0.07267464697360992, 0.2165430784225464, 0.3734228312969208, 0.3373594284057617], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "650d964bed3a980d90400ff04a0bb4afb82afc4627d1a6a931c9f249c7b3fb3d:action", "state_id": "5d415b6faf5fb3d24bce4d8c8621fff3b104e199312eda4a02613dcc317c6586", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.69677734375, -3.37890625, -0.130859375, -0.18359375], "student_probs": [0.22221817076206207, 0.015203576534986496, 0.39134031534194946, 0.37123793363571167], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "61c96254baa9621878ba7343e73f694845cf8506efcfbfdff7c4f26dd6190967:action", "state_id": "0eaae98f83f8b485fad71f841d21c8e0184fdec656176af15c64bfbd2897f31d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.828125, -0.60040283203125, 0.8203125, -1.03515625], "student_probs": [0.04818039387464523, 0.1644611358642578, 0.6808823943138123, 0.10647613555192947], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "138817733244265c649e95b49bbb56aaf9d4411ec920a8329ac656eeba845be8:action", "state_id": "feaedb0abf309820772603965faa790e50c30cd634244e983bf072ceaa71f7d1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.703125, 0.06640625, 1.36328125, 1.33984375], "student_probs": [0.1867627054452896, 0.09880221635103226, 0.3614034950733185, 0.35303157567977905], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "af12acdd5d39c7618670e7f9047efb0b37756dd526550da63d9402793a2423f2:action", "state_id": "a08b18838f96dd88df1122b05c47c761febf994dd4458d31e9440d17130cd61e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.80078125, -0.154296875, -2.5390625, -1.3359375], "student_probs": [0.12108873575925827, 0.6282938718795776, 0.057872503995895386, 0.1927448809146881], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "35725207b352a92ba5fbc2f42d198c09fd9b09d5d5085d4bc4745959e92fd0ea:action", "state_id": "db19e579b7813c6aa448890d23d910ac92871c0cb1964a6ead6cc6ba93d4d567", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0546875, -0.103515625, -2.31640625, -1.66796875], "student_probs": [0.0972876250743866, 0.68460613489151, 0.07488495856523514, 0.14322124421596527], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "29da5a5ad7b64f85d9d15b8ca8ebb5e8791660229d43d6a991d553e895ddd832:action", "state_id": "a44431945d4b36447481a88e4e397272253027612ebcdd8afb53f90dc823da29", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.484375, 0.296875, -2.2578125, -2.046875], "student_probs": [0.050144683569669724, 0.8092942237854004, 0.0628955289721489, 0.07766560465097427], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9e1aedce502bdb86bc55187e3cd3bf28f42b949f813b7db895665bb7bdb166e2:action", "state_id": "bbca06445908ed09602a5ce9431d1d8a7f6991403793835523f3fb1a48ca357c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.96484375, 1.5859375, -2.01953125, -1.390625], "student_probs": [0.02593155764043331, 0.9034691452980042, 0.024551507085561752, 0.04604777693748474], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0b314f089fcba1c176ab530984349d9ec07f71043f5bb6034a2e7f17e3167da4:action", "state_id": "134447588b5583e1a6622b1594204117cc813867030351c15cc596e24c1624c9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1171875, 1.5234375, -2.06640625, -1.38671875], "student_probs": [0.023672115057706833, 0.9022780656814575, 0.024905258789658546, 0.049144573509693146], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ecf208e912670297b2230edf7e19b8f1113e427d8532d206aa55677fd79e6ff8:action", "state_id": "155ab3ca9cd432977316286f842a3be4c4c1342578039b717d24ad1dca3880dc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2890625, -2.89453125, -2.0703125, -1.130859375], "student_probs": [0.16737675666809082, 0.09135733544826508, 0.20830373466014862, 0.5329621434211731], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "068c059e59e4663210106afd6eea6d05545752d955eac0a9a911e68711f1fd58:action", "state_id": "accee77de200c9845130034b0f4f5ced9d36f022549c1107941a023580ec5bd1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.84765625, -0.5711669921875, 0.71875, -1.060546875], "student_probs": [0.05050499364733696, 0.18101166188716888, 0.6575221419334412, 0.11096131056547165], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c5fe9854858aae1ec816822fc0e557f6108393159b2c21e3ae29d172a8cfaebe:action", "state_id": "a55228609ff5ccb61c5eea95815ed7cdfe47459466e608e79ecc7eab45a40a13", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.697265625, 0.14453125, 1.3203125, 1.333984375], "student_probs": [0.18760870397090912, 0.10794524103403091, 0.349815309047699, 0.3546307682991028], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9b7f837111ce45b1c53864319a6ffa11b86a98cd4202b645f8772f33d5a90771:action", "state_id": "2063a900045c2c3cf321656712b2f6ac349e8afc012976324621ddb999d823fe", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.01953125, -0.251953125, -2.7265625, -1.4921875], "student_probs": [0.11056829988956451, 0.6475600600242615, 0.05452188476920128, 0.18734975159168243], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "657a5aba9c85f05d7f20e3c45e3f1c8acf65e11352d9e16d83f2b8964ba33c75:action", "state_id": "2568c4e6e9ef86f0ff7cbec6b4d8243ecd1b05bebf0bdf46fbcb9e29413996ed", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.21484375, -0.21484375, -2.79296875, -1.96875], "student_probs": [0.09776102751493454, 0.7223617434501648, 0.05483897030353546, 0.1250382661819458], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "70ba775bd8a01d98e05c3bbc6552a6b7052bcd16d843786315f9001154414b27:action", "state_id": "9b752aaeec475b8a34b7fdf7253afcda52e94d8a75312152672f170923f357c6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.41015625, -0.55029296875, -2.74609375, -2.1640625], "student_probs": [0.10619605332612991, 0.6820822358131409, 0.07589489966630936, 0.13582681119441986], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fca07441bbbcfc517ea4cf88a77c5638096c8d2379de2ea1cbacbcaa8b39b841:action", "state_id": "8427dd95fe24982009e04cc65d82ab064400dfec4bf07371cd6d0f123bbcc420", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5546875, 0.6953125, -1.48046875, -1.1171875], "student_probs": [0.0762566551566124, 0.7235029935836792, 0.08213164657354355, 0.11810861527919769], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bf409a84ecd9d0a574244e1cc23c3ccdf9954bea4cc0d91f999bde621c5f055b:action", "state_id": "a847437c2a5feb4a7b52b01d543071cc6d78efe7f858a1833de078682b64b278", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.66015625, -2.69140625, -2.2734375, -2.21875], "student_probs": [0.20014940202236176, 0.19399145245552063, 0.2946484088897705, 0.3112107515335083], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "817a51b5ecccdcf4fab619f2f301425eb4119c33a270ef77048390f1ee6ae405:action", "state_id": "5fc466c6ee31ce1356eba20da917062d66fcbd09e9d3ed3c9c4ee077a8e64ab8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.162109375, -0.641845703125, 0.9921875, 0.18359375], "student_probs": [0.16119354963302612, 0.0997701957821846, 0.5112724900245667, 0.22776377201080322], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2c80fa520c0a9584dfb3fe6cb42ed71afb761cc8c4dc9cc483215fc701695e1b:action", "state_id": "796edc31a75dc655a6de433c71b40a9a71cad2777647267608bfed7b557e843a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.025390625, 0.21875, 1.826171875, 1.5234375], "student_probs": [0.18800033628940582, 0.08391489833593369, 0.4187294542789459, 0.30935534834861755], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "74db57fa431e7a779b22309399d30d942d7800b5548335238e2013cd8e0af6f5:action", "state_id": "91acae17d83216e6f529eb7a0c8806e8eb3d7f2835c0b0a937fb4ed78e92c5bc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.41796875, 1.380859375, 1.373046875, 1.3515625], "student_probs": [0.11413226276636124, 0.2989417016506195, 0.296615332365036, 0.29031068086624146], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b5152c48e7970905c13068fa21dcfc4666f483d5cef31b8425ce555a32d1bed6:action", "state_id": "6cdeee570b8265d6a53a3a1586cb77701e27897e665d19f4248963b1b815a4d3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4638671875, -3.0, -0.08984375, -0.15625], "student_probs": [0.2568763494491577, 0.020337408408522606, 0.37338805198669434, 0.34939810633659363], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4c24f27adc5fff55a2297ce4b60e806fa945c888f312cd701bbb540753c3dba7:action", "state_id": "cd173c8c8a1875a7e7220b96e177d5ac6b91a3cfac619f92b41b8f250c6771ce", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.578125, -0.19921875, -3.1875, -2.7265625], "student_probs": [0.0757642537355423, 0.8177305459976196, 0.04119231179356575, 0.06531286984682083], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "df26da9952c59a1733fc2caf20688e2a4d56cfc9642e4767573399712c907953:action", "state_id": "0bccd7374c95b4b66a6efcec2908e78f8dcafde0fb65614ff3f970486650d75e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1953125, 0.33203125, -1.34765625, -0.7470703125], "student_probs": [0.12453025579452515, 0.5735771656036377, 0.10693326592445374, 0.19495931267738342], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4c3beca7a69d42902193fdc21f470a4821710028b0f45ca3f7db52c2ffba3c6d:action", "state_id": "53b05fda4264621b631838a8c164a2efa6e6a82e628b4cf15fcfdd934917c021", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -2.04296875, -0.82275390625, -0.9462890625], "student_probs": [0.2706654965877533, 0.09879738837480545, 0.3347172141075134, 0.2958199083805084], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e75c657a98a3747e779d6dfe94c7d789d3771239d3b966d30a8a2dde10aec88c:action", "state_id": "a421e79ab32b973a434909ed056f4cbf710bcc200bd6572ac02aca4f06491ad7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6328125, -1.6796875, 0.96875, -1.013671875], "student_probs": [0.05781573802232742, 0.055168163031339645, 0.7796331644058228, 0.10738296806812286], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9dbad0bf9a0c3bc3d6b7b9a9bd9645a9d92743f0adb34ce43e7a51503541322e:action", "state_id": "fcb1cc89653197eaa67f9d07c39a91f00dbf9fccdca0637ed1bcf4ca3971409f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.748046875, -0.04296875, 1.498046875, 1.26953125], "student_probs": [0.19029821455478668, 0.08627818524837494, 0.4028613269329071, 0.32056233286857605], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ecbab0c00123a482aebb3748d663842b472141382d7f627aac9341ad13363ff7:action", "state_id": "7ca1df2804b0b9e3dad862d6280d3c102743fa847a9b3d4bc821819c0f2ec1ac", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.921875, 0.47265625, -2.109375, -0.688720703125], "student_probs": [0.06163661181926727, 0.6757256984710693, 0.0510985404253006, 0.21153919398784637], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c8a52e92fea0d9a35fe83551b3a84c3c1b074992541c528fe0d85ebd9a3e79f3:action", "state_id": "5fb34d498930b7995cc861b6514a0f831d35c1eb47b90e09040fc83d5027b5e9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58203125, 0.85546875, -1.26953125, -0.4814453125], "student_probs": [0.05946307256817818, 0.6805188655853271, 0.08127638697624207, 0.17874164879322052], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3a613de65f94b0935590b45574c376e359a1bdeca7f3c183d9d8f1cf948fb263:action", "state_id": "1122b6e4fb84526318a547fae10d1d43cc98a7475081394063a26af957d4301a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.109375, -3.13671875, -0.7509765625, -1.1640625], "student_probs": [0.1278521716594696, 0.04576552286744118, 0.49733966588974, 0.32904261350631714], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0b8057426a01c438e3bcc61d8b3067df843bd1494ea9a7079bd58d44c1687196:action", "state_id": "499590d022aad8cf242c52a57be3f7e6a09b87d0ea60238caeca74971a72c2c2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.12109375, 0.796875, -1.9609375, -2.23828125], "student_probs": [0.01757275126874447, 0.8838772177696228, 0.0560646578669548, 0.04248546063899994], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7b745b350cedc5f11d248dc276fbf24563ea4eccbb1bd1b390ca98d63fb49c90:action", "state_id": "a3f0db9df62a75022a208180dae0585fcfa0d28d1b04f4834d94e00f53142110", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.984375, 0.1953125, -2.3203125, -1.55078125], "student_probs": [0.08263778686523438, 0.7308107614517212, 0.05905856192111969, 0.12749291956424713], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "902528d0d7c52adb2bf4b90c538cea5219dc0458ab227af0fe1b83e71cfc4451:action", "state_id": "c7792d4b48799855d4ab75949849ce4263dc971dfcd68911b3db5630f864db6d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.734375, -3.4453125, -3.53125, -2.5], "student_probs": [0.31191152334213257, 0.1532057821750641, 0.14058953523635864, 0.3942930996417999], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "22337f799399c118c5801aa33b07b1a6be1754aab186de2fa722c2e552d442b8:action", "state_id": "78a3a0f0a1337e493c37e51c41e8f9f6c899a1e92bcf74398060e38feb254042", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0546875, 1.0859375, -0.447265625, -0.95703125], "student_probs": [0.031147431582212448, 0.7200760245323181, 0.15542350709438324, 0.09335300326347351], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aeebd23c8a0c6bd9c7e63593c7c6236c40dc3e082684e7782269ce704537df5e:action", "state_id": "9a6eeade6d38e63317f6a7c4d8cd8da926894338aa0824225239b23940cd29c7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0263671875, 0.6171875, 0.30859375, 0.49609375], "student_probs": [0.06869611144065857, 0.3554011881351471, 0.2610347270965576, 0.3148680031299591], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1e5d57d17d37d7d9f07c08a3b27fa417b35f85f5659997f0a0bf71890643caed:action", "state_id": "3bcc4752d0ad73cf23d6137adc99bbebd3ebce200224ca3b5bbd7cad0555af96", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.80615234375, 1.30859375, 0.4765625, 0.609375], "student_probs": [0.058780115097761154, 0.4871390759944916, 0.21198560297489166, 0.24209517240524292], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6bce03eb4c40ff251f553c472786fa62e748a51986642197eb4c3417bdd84b48:action", "state_id": "962dedbe7da188ca8c00543e5f228fd5139f53a4cc7e9bbe7bb5066884fed7ad", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8720703125, 1.73828125, 0.390625, 0.46875], "student_probs": [0.0455353744328022, 0.6194556355476379, 0.16096465289592743, 0.17404429614543915], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b921de6d8df7c221b533f8a9882291d7e61f81e7319d5a038380ea657e45a420:action", "state_id": "d387a5fda66f6d72472cee86f8c2d1ed2cff29c27dc705647a14bdd150b72be5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.16015625, -3.62109375, -0.3857421875, -0.52490234375], "student_probs": [0.19447015225887299, 0.016598979011178017, 0.42186811566352844, 0.36706268787384033], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3cd5dc74c5b79939ff21b50580cf3471355a72248af0ef96fbefb75eeb68513f:action", "state_id": "fbcc0db7eaa0c3e35123c1387986f67c34948bf90192db2286113ffe5fb92b97", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.53125, 1.3828125, -0.890625, -1.16015625], "student_probs": [0.016611263155937195, 0.8322587609291077, 0.08568740636110306, 0.06544267386198044], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "77138aaafdd44fd6cb4a715102c99fa1c469c79ffc6900f3ac5b13ad4ff633e9:action", "state_id": "4ee723a610a8225f597964d69736de0582286626cf7a36ecae4e54b02e5cf14f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6484375, 0.953125, -0.07421875, -0.08984375], "student_probs": [0.04155603423714638, 0.5603744387626648, 0.20058968663215637, 0.19747982919216156], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f81ded58fd3e74c356c440e95d24f5af02445e5943aa1d4e0ca8c14127ab7105:action", "state_id": "8a624272d80719df17da95afa5e1404da3c1a82f2d8b9b47483574ebf602ae6c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6328125, 1.220703125, -0.59521484375, -0.28125], "student_probs": [0.03994479775428772, 0.692988932132721, 0.11274132877588272, 0.15432502329349518], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c91dfdff754973c4c24cb5af7cb23da0147ef7d68148ac27e35f628d08a00645:action", "state_id": "c3ea4de6c08ff0bb9c75a7188ed4a73e66ea28788875960713052b52f1a69ece", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.64453125, 1.3671875, -1.146484375, -0.53564453125], "student_probs": [0.03846333548426628, 0.7816634178161621, 0.06329157948493958, 0.11658168584108353], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "86bd5f8fdbc02efea0f794edc63fa7d2b8da93b6b686927e197b57e63d9c5a8d:action", "state_id": "47b4f68e3588e80e270af694938f4331a627502e0536ffdea73ee6169edac68a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3515625, 0.97265625, -2.23828125, -1.7578125], "student_probs": [0.031537774950265884, 0.8760339617729187, 0.03532063588500023, 0.05710754171013832], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0f818a3ee67969a3793d7f1725a5aca16d9fa8c8ff0366cc6280ddbd8c9506a1:action", "state_id": "b6c63c11a4e7da1f1daa60aa104460447bc84aed5e0c566e30dfd8c9143fbe33", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.95703125, 1.70703125, -2.09765625, -1.1875], "student_probs": [0.02323036454617977, 0.9064381122589111, 0.020182890817523003, 0.05014864727854729], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "385029f8f5cd780d83c7c81284ff9876ea72cd8a4c55513705277e50499f376c:action", "state_id": "d3036b4e771d1dc2354003ac0cb16b95d219c2c131708e17e632419dba39812d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8046875, 1.791015625, -1.2421875, -0.5992431640625], "student_probs": [0.023510266095399857, 0.8567449450492859, 0.04126180335879326, 0.07848295569419861], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d4080f10c8ae9292fc83c26fa68dfd39aaa5e7b8fe9a3478d7daf55513e6cad6:action", "state_id": "ee1ee27bc4c0c52b9346f728d055774fb7a270a4d03a7e3cfa0092c7ad9e6f1b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.47265625, 1.8125, -0.8515625, -0.337890625], "student_probs": [0.030595483258366585, 0.8173019886016846, 0.056937046349048615, 0.09516555070877075], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1616d777a92ca7af7750b70224099641bfecc9008e73dd966609c75b672b1c55:action", "state_id": "56b05919f2ca82f239d9fcc901c77f236dffabe926609bdcab34a8487f9c7cca", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.734375, 1.560546875, -1.3203125, -1.28125], "student_probs": [0.032194215804338455, 0.8684488534927368, 0.048708293586969376, 0.050648611038923264], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "acd26f40202e76c05e36d26403947cf874e64bff1139e593303c564081dbb996:action", "state_id": "f4e2d0b1c8a0acab2eaa4dfbdc3f8fdf850fd0afe6b7a075494223da81751bf1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.359375, 1.826171875, -1.234375, -0.81201171875], "student_probs": [0.035660356283187866, 0.8622855544090271, 0.04040847718715668, 0.061645664274692535], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3e629c930113647762027250b9bfc356a0e1b197647a981db6256888526db3cb:action", "state_id": "45f3a73a53bc97dcf793ac77296b1eb766421a4675aef7615d3a9f531545bd14", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.71875, 1.546875, -1.8828125, -1.181640625], "student_probs": [0.03360641747713089, 0.8803697824478149, 0.028521394357085228, 0.057502381503582], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1fe5c362ef4e76c16c8ca1d8639705391a0f7e3f001d25082427e83563560150:action", "state_id": "6dcb4620a1e0d42ba4c5c54a1da30eb427618b0015bcd5e91a594662ef4d6056", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.765625, 0.3984375, -2.3984375, -1.3984375], "student_probs": [0.08560764789581299, 0.7453374862670898, 0.04546587914228439, 0.12358906865119934], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "59f2bfe5b89641b7b3143344e40db180daa0f9793190655892e5f2afd1bba1da:action", "state_id": "4a53f8afb2d69f00058905223caa58672c74b24e09866fb99926f02ba00d3a1c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.28515625, 0.8671875, -2.171875, -0.974609375], "student_probs": [0.08786436915397644, 0.7560731172561646, 0.036200594156980515, 0.11986201256513596], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b5c1b865fc7a70087ce04da9149c288fac5bc30cc835add96fed10cba7cfc812:action", "state_id": "2dc7a0cceed4738ebc0c4ff088859a61d8e0ffd614dda90a4937dd1d205f17f4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.46484375, -2.109375, -1.6796875, -1.15234375], "student_probs": [0.2703861892223358, 0.14192800223827362, 0.21811170876026154, 0.36957406997680664], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "15c8448e81afe15dc53d46eb84205515325751c6079c0a6a65b0fd6804d2be66:action", "state_id": "f5c868d837bc8e448c8b88417b0c4e01ded33d0a09a677860b1c9de34fd016cc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9619140625, -0.8388671875, 0.662109375, -0.015625], "student_probs": [0.10224393010139465, 0.11563149094581604, 0.5187307000160217, 0.2633938789367676], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "33384a2200dda67ef3abfa2ff332f55a621b2c0dd4e67fedbc65d81e6ce75bf1:action", "state_id": "cef1c22652e7cf2bb142c8593e4089fe9d8a3d5d16cd68af169db0d220688b4c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4765625, -0.44140625, -2.06640625, -1.23046875], "student_probs": [0.17702314257621765, 0.49841681122779846, 0.09814409166574478, 0.22641605138778687], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "978ca8fd60a93d8ccd4867ef2fc7faf837818f5977c03f9bc609b6f945839217:action", "state_id": "b609830fa181721151f6c1de924e2f80365c786aa4629c9699b409fb23dbcbb8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7734375, -2.6796875, -1.99609375, -1.50390625], "student_probs": [0.2845940887928009, 0.11498639732599258, 0.22778621315956116, 0.3726333975791931], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "db92d4c95b98afcb97eb267148488c05a367a3e1ef9562d072086a4d94016a6b:action", "state_id": "1eadf507a73346feeedfa8c2910c127033bfac663cf75f0a401d0fd636c368f4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.34375, 1.1640625, -1.0390625, -1.5859375], "student_probs": [0.02487851493060589, 0.8303249478340149, 0.09171556681394577, 0.05308089777827263], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f254ed142b0cf639cff8fb5149de4e5a1ae2995bd8cf08e1f267ad300d373d86:action", "state_id": "8820a6f7dade521b52704f4dae16767084263b347046cc3fe4e468b385338a61", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7734375, 0.73828125, -2.140625, -0.8056640625], "student_probs": [0.06005697324872017, 0.7402681112289429, 0.04160024970769882, 0.15807460248470306], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1b044a93b94e422cb26425adfa1d672c26d2025dee5ffcd1caf72976bbfcb598:action", "state_id": "74f2713fe777519d7bcd779d40389a2d526cd8f4cd21d65abab94297c30d258c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.72265625, 1.23046875, -2.14453125, -0.5931396484375], "student_probs": [0.04181348904967308, 0.8013870120048523, 0.02742195501923561, 0.12937764823436737], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1b89d8e415915e0d1dd5f929810d1baaaced6a6219c7a144fc42f532dec9742b:action", "state_id": "5b1a22a06b9c6436300c7e449d9d11d2a15cbbd2924b81c5389dd71d04e687ad", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.4765625, -2.8828125, -3.375, -2.3203125], "student_probs": [0.30840709805488586, 0.20544344186782837, 0.12558504939079285, 0.3605644404888153], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ec614601528de941c5113fb57349ffc090830d38b122229e9498733d19a3b9d8:action", "state_id": "4b243bc278d2c4ad07eaa5075d7c6212910b6c6e20491cc4ca0003d2a7e8987f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.66796875, 1.19140625, 0.0703125, -0.50341796875], "student_probs": [0.036572884768247604, 0.63821941614151, 0.2080104649066925, 0.1171971932053566], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0355ab2084307ca1ac9b5cd8a92cdb7190bddd4de215652410abe16b71c8525b:action", "state_id": "5b5b33446ce9793bd58132a25125cc1ae227459caa71c397c51fd8b237dc7541", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8671875, 0.72265625, 0.3671875, 0.484375], "student_probs": [0.07574225217103958, 0.37136295437812805, 0.26026779413223267, 0.2926269769668579], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e52639d71caeeaa2876ddaad9b50b2aa24f13a355adcb8b3be8cc4108d2133ea:action", "state_id": "2d208261e3082625e234d3d04ca8ad292b0cc90940c8e0e73a9a1cba6bd79295", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6220703125, 1.556640625, 0.5390625, 0.728515625], "student_probs": [0.05921313166618347, 0.5231426954269409, 0.18910004198551178, 0.22854411602020264], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9bfd75f1814c91bf76a44a2aee8ba419805cb66e748701fd33d5e171fd64d83d:action", "state_id": "cd04577ba4da9b9ca43e2027c76cd6e5d5d09e5b5160223055abb42cdbac64ef", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0234375, -3.359375, -0.298828125, -0.421875], "student_probs": [0.20057713985443115, 0.019399773329496384, 0.41397613286972046, 0.3660469055175781], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "898326c11a2229c91f509d0046f0d2f1e4e5583ec3cf37864307492a93884753:action", "state_id": "ef8864584fc3e28407377e1226d2bb7bee1211f6d3c86e19c7221e2aa80f06d2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78515625, 1.185546875, -0.00390625, -0.62127685546875], "student_probs": [0.03373223543167114, 0.6579684615135193, 0.20027749240398407, 0.1080218181014061], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cefc64b24d04cf7ce6620ee768ca5f4dcb6e2a57ac2792ed4c5a99b95f15d98b:action", "state_id": "01c2fe243eae20bf27ddbda7e42e7c9c277b342694d3c62a986e41d4b0878fce", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2578125, 0.6484375, -0.05078125, 0.1875], "student_probs": [0.0652974471449852, 0.43930894136428833, 0.21832486987113953, 0.27706870436668396], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f1a79cd30f13f41b53209017f23458cfeb974187f818b99137cc4d5c2604c8ec:action", "state_id": "e1ecb3d7218f842ea744c183c55c387452af8b1833add85c911dffd57328bf51", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.296875, 1.427734375, -0.205078125, 0.1953125], "student_probs": [0.04223528131842613, 0.6441072225570679, 0.12584522366523743, 0.18781234323978424], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "af9094a390c384363b25aa558d024ab255c0dabb9f5d10fd83b9d788b48a1b71:action", "state_id": "ec11e6151fb6f78c27c2ae90c4794157cf2f8643c7ca8a4765089faff8755cf6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.359375, 1.826171875, -0.27734375, -0.0234375], "student_probs": [0.03131386637687683, 0.7571852207183838, 0.0923967957496643, 0.11910416930913925], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "52c2980abe8746dbc8a6b87a5c36f31b22aee342c0333da21741800c28d71f1e:action", "state_id": "5691b3509cd20cbe456be9a4ef9e10102b95e4a62c425eacf739435fa0073f66", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.70703125, 1.849609375, -0.828125, -0.663330078125], "student_probs": [0.024217039346694946, 0.8486926555633545, 0.0583210326731205, 0.06876931339502335], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "414cf5cdda7d8d6688d715cf3551f184a4da41b1ec3ccc32ba8fb91bbf5449dd:action", "state_id": "cf4f5fd9516650a591e4beb47e5f324a49c3bb93c2156a9f01e1148b389447d1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37890625, 2.193359375, -0.365234375, -0.04296875], "student_probs": [0.023171525448560715, 0.8248403072357178, 0.06385380029678345, 0.0881342887878418], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6d274562c2a4c9b764173fc68d439a24e05db1c78dc5eb1208c889778cf5ad6e:action", "state_id": "5cd5af4882cb23f270b1e581bc28ab20a912f595beb99e7b5715192d9798090e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3359375, 1.92578125, -0.55810546875, -0.078125], "student_probs": [0.03049820475280285, 0.7958307862281799, 0.06638690829277039, 0.10728408396244049], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "56b204f60a85b96bbe60feddbe7e56b65550c7b372f6470e24086b5bf764b1bc:action", "state_id": "30fd64dee1d79cb0297eaca5e59bb2f82b37adc2414318197980aeae8dec1c65", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.255859375, 1.990234375, -0.5947265625, -0.0546875], "student_probs": [0.03129813075065613, 0.8040425181388855, 0.06062402203679085, 0.10403530299663544], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7b177456bb6956098d8b437f3f3b98e32b2b7634117dd869023fb51792caf670:action", "state_id": "1225152ad53d7246019cbc7960faabcf969a2d31609df04b9d6784fd38eb8253", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1640625, 1.6796875, -0.83251953125, -0.490234375], "student_probs": [0.046436209231615067, 0.7977773547172546, 0.06469102948904037, 0.09109543263912201], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4f4a1ebb3fb9898523c19a188a100e36bfb62516cbd36228f8ca61e6d6d6a419:action", "state_id": "576966701ed333bae6bdcf3504f400974228d0ce230917737b06e48cd54336e2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.09375, 1.783203125, -0.667236328125, -0.341796875], "student_probs": [0.04461672157049179, 0.7923964262008667, 0.06834868341684341, 0.09463825821876526], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7e27d499c45540af56f921eff1ac8242b81876a0982ac9df9f288ae1a1dfb4ad:action", "state_id": "028a8f96e2ad3e45533b7b274488eeb9e6ddb9ef0df77ee2e5ec7890a6835b7f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9541015625, 1.9609375, -0.52734375, -0.1015625], "student_probs": [0.04286802187561989, 0.7908949851989746, 0.06568587571382523, 0.10055101662874222], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c37fb678f882d2513a951027db7b3f498bbce2fde503a75a64f2b96463402863:action", "state_id": "4c964d284b110d1daba65e78e2425153a84108e76e9e73d356593c8dae1059a4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.07421875, 1.705078125, -0.33203125, -0.109375], "student_probs": [0.04580307751893997, 0.737781822681427, 0.0962105244398117, 0.12020456790924072], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "966ec4396a39096d51db859775c710d7b3826ea3be5d607949d192d9452e8264:action", "state_id": "66d53ec64f1c04df1380250cdd1dad56c1198b4a744ba1e60e570fa1683013cb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.357421875, 1.013671875, 0.44140625, 0.65625], "student_probs": [0.1008237898349762, 0.3972111940383911, 0.22412468492984772, 0.2778402268886566], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "259a3b85097f0f4fc8c8dfd9b1f9082df2295a217feeb7067a5531e0477beffe:action", "state_id": "54a39637deb2f6e805c39c8c9248495c61e39ba54aa23987e45aad248abc5175", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5445556640625, -3.03125, 0.02734375, -0.00390625], "student_probs": [0.2187257707118988, 0.018194593489170074, 0.38750091195106506, 0.37557876110076904], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "af5da9e90cb1bcb92b4df85bf62a3c852deb76a01e9a8825da343bd29e6e2f59:action", "state_id": "de4a4a7981d4e27d7f160092184fa2cd2a4c92c2be2f2129dfdf103b9aa426b8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9765625, 1.302734375, -0.8154296875, -1.375], "student_probs": [0.030697811394929886, 0.8152446150779724, 0.09803495556116104, 0.0560225285589695], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c58e15f276c770153d9f76e6183adf54d29626a33babeb8f1f56040f478cef50:action", "state_id": "00033b2bd5e36f9466e8c08733a78a2b926e8a9ba6ad0520ff3065a8266b60b1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8031005859375, 0.55859375, -0.87646484375, -0.015625], "student_probs": [0.1245344951748848, 0.48603329062461853, 0.11572521179914474, 0.2737070620059967], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "801b44e9eef20b16ccd8ed8bfaaa10c55ff625da0d67813104c60274dca0fe06:action", "state_id": "933397ea0f63d08eb4164047f094df71abb26e37d7e8af268ec6e7b90b70f97c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.72265625, 1.08984375, -2.04296875, -1.072265625], "student_probs": [0.04927636310458183, 0.8205251097679138, 0.035770803689956665, 0.09442776441574097], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "70f232b24b845071483e12ed159de37e95c60f93eac0e99fd8c498d33ebd1e48:action", "state_id": "14301cbba2fd49cc5e142d15565cba5f1c9c429ddca6cac05b7432d4dd055cce", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.64453125, 1.46875, -2.0546875, -0.585205078125], "student_probs": [0.036978546530008316, 0.8318225741386414, 0.024536987766623497, 0.10666190832853317], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "085af15d13f74a02946dca428209950af39143d3a1de2e418195123a2c94ff2a:action", "state_id": "a2d2e0d3cb19f09cf2c2e834e38d090ee7f6fd33e09ff15b618e53395519869b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.08984375, 1.95703125, -2.2265625, -1.546875], "student_probs": [0.01644420623779297, 0.9409106373786926, 0.014342891052365303, 0.02830226719379425], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5afc1e3b6f1d4bd97c6c29131e278a511c9d138ce03454abe723d9a340fd9d12:action", "state_id": "b7637e7bacdddd5c82e1aedc5e320e3e44846d0cbf7fe1f797133980e6de8bb6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.140625, 1.82421875, -2.19921875, -1.5703125], "student_probs": [0.017722971737384796, 0.9342139363288879, 0.01671435497701168, 0.03134874254465103], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e4c9642a30cac08c9a0bb004715e632f69a6c11d947ff8b30e70ea136e26fd98:action", "state_id": "b6a9795ae3ec65873fbe098ca11744c242345adf7ea49b6d96dc9e96477b9a98", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.17578125, 1.734375, -1.828125, -1.65234375], "student_probs": [0.018514983355998993, 0.9240226149559021, 0.026212504133582115, 0.03124995157122612], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a1398ec1089302149b0c68e2b5d833a1f017613040e677fca32ce3d172a941a4:action", "state_id": "e9e0e7c105bb1b264b1716cbf3bccf98a43fc54aab935a0143ad5301f4bc50f2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.12890625, 1.40625, -2.33984375, -1.84765625], "student_probs": [0.02671298198401928, 0.9162652492523193, 0.021632831543684006, 0.03538895025849342], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b18a49ad5a3c4ef989d14da5e6bfd8e3885cbabd1268019c22db69e37580e184:action", "state_id": "e65ff7193a24c81048228ce481273480a2ed72671191703d1a8149a47731474f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6015625, 1.015625, -2.2578125, -1.42578125], "student_probs": [0.06094544753432274, 0.8347786664962769, 0.03161807730793953, 0.07265777140855789], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fb2e5cba779962f361a58bcf21090222ac45cef60ea8ad48abe00d9f11b9aba9:action", "state_id": "c0471c8ab1aae68e9d37596c7122756391910be5657dfbf9887531cee27ddf1e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.90625, 0.71875, -2.70703125, -1.52734375], "student_probs": [0.059829231351614, 0.8259170055389404, 0.02686201222240925, 0.08739171922206879], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "21c4cb17e414950fd039f269d5def3f64c67bf795b2f17291d180ef4e164f0bf:action", "state_id": "4397c03904b618218da778b5d065d1f12741883e7564650dbdf97a8bd369a3a8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.296875, 0.671875, -2.72265625, -1.99609375], "student_probs": [0.044500332325696945, 0.8663133382797241, 0.029070252552628517, 0.06011611595749855], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e0bc0d4d640db4baeb08eae63315a561540c681b6d74337a1083a2fc704cd886:action", "state_id": "adbbb10ff25b5a6b02000355d12c261155e684fe6509b1cc871872b7a9db6e49", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.27734375, 0.79296875, -1.9765625, -0.97265625], "student_probs": [0.09276030212640762, 0.7353386878967285, 0.04609939828515053, 0.12580162286758423], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c5e7588904868818113b2081027540cbf3076ce666c0f8e8bd99abf93d8fd8c8:action", "state_id": "26ec5795bbee8020cc48c09429fdf841241e8af40793efa3e0b72c1dce95f6a7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55078125, -2.09375, -1.85546875, -1.23828125], "student_probs": [0.2713547646999359, 0.1576627492904663, 0.20008444786071777, 0.37089797854423523], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f84c87a41a8b8e9bec76ae920652c81d703f852fb7c3bc5967c8912b62b94fa2:action", "state_id": "3c5bb8bfc0efe84cb0aec8ed7d8a31fd3ffa83e160eab70fd96af1fafa894dd4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78125, -0.75048828125, 0.849609375, -0.9296875], "student_probs": [0.04991962015628815, 0.1399346888065338, 0.6931687593460083, 0.11697691679000854], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "60a1e4b39991c2d68c2519efb68c67058349e6377acd75ac839273c9259f046e:action", "state_id": "0aabfaf6e7af4cbeb9c98c5f4180fdce63ee182c958aed1220df5ef28cdcd001", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.2265625, -0.10546875, 0.673828125, 0.5546875], "student_probs": [0.14763426780700684, 0.1666393280029297, 0.3632635772228241, 0.32246288657188416], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dc67e423c8a3df799aac69f25c19f723845454c270e2fc20247a402db0117036:action", "state_id": "010fc32e25ca73ef894d98d2080d8960931435de3c9bb1d10c1c5fe820cdeb05", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.84765625, -0.349609375, -2.7890625, -1.4921875], "student_probs": [0.13717660307884216, 0.6135833263397217, 0.053509701043367386, 0.19573034346103668], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "90bed11635066c184cb968ed53ee3aa4b24171410f4ff1a74edb65dc9b0a567a:action", "state_id": "7a32959e70556f01d319678dc8e65abb2553a7b9fb818c9ff8d1e275af555bfe", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.25390625, -0.4169921875, -2.75390625, -1.98828125], "student_probs": [0.10883863270282745, 0.6831950545310974, 0.06601396948099136, 0.14195233583450317], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4aa437abef3767be1c0e4be2ecb7719ba570ac8ecd616eceb3b49b05fb50d8bb:action", "state_id": "e52b3600cf0554340696fe2d1ee1620ee654c0b44187b4cf383f2cdc7c8bb8b7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.265625, -0.115234375, -2.796875, -2.1796875], "student_probs": [0.08876406401395798, 0.7623246312141418, 0.052181702107191086, 0.0967295914888382], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f5830eb9591d8b91e520614143a0c8cb0f9c9be5943ffcd3731204180974d187:action", "state_id": "399f4470ddc605bb761d0681b5d59a78868d12f1a8b605376454375546c65cd6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.18359375, 0.11328125, -2.25390625, -2.16796875], "student_probs": [0.07757403701543808, 0.7713234424591064, 0.07230695337057114, 0.0787956491112709], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "141fa7a25f1fba57ce2602d8dd99b23fd287b92da22af308066353a6e1544c09:action", "state_id": "a33f5868774fac8ae15c2f7982771718ae5fa23c6f57eac4b6dda850a7a62840", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.05859375, 1.048828125, -0.771484375, -0.9599609375], "student_probs": [0.08573950827121735, 0.7053792476654053, 0.11425389349460602, 0.09462735056877136], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2cec9c62c920527e0851d0e685ab8782bf2ffc711a9d60dc24f5e34a279542fa:action", "state_id": "466f6cf945f776a1d9771c971c22daf3afcb95974c7a27d1c395e6ceda9c6fdb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5908203125, 1.578125, -0.6611328125, -0.109375], "student_probs": [0.0813036859035492, 0.7113301157951355, 0.075783371925354, 0.1315828412771225], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c6568c15af9891284cb1cdba4277d47d9a903339c71785212e8dd2a148190b6d:action", "state_id": "ea09bc25ec4d5b89d7d12088293f8d07295136c4868e8bd0b3a45ee5d46cdf1d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.890625, 1.361328125, -0.9033203125, -0.45703125], "student_probs": [0.07670792937278748, 0.7292073965072632, 0.0757402554154396, 0.11834437400102615], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e71d1fb8233d762f0a15b4227a7c7c5c619e988e8c9e374619611735226d85f8:action", "state_id": "f393d127e3d948d92339de1b8a3e2de8adb3b58d3a93f5a51f032b5a5d2c3ea8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.82080078125, 1.443359375, -0.6552734375, -0.314453125], "student_probs": [0.07428165525197983, 0.7148153185844421, 0.08765348792076111, 0.1232496127486229], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "403dbe3ae292876b306b0a26b26bf5c2aec99a3a0eaa359c8e8962a72add53e2:action", "state_id": "db4383f968cd92716a8cd07c3f97d7fbfabf16fc2246dae96f393ec30fb47c84", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.33984375, -2.48046875, -0.9052734375, -0.9287109375], "student_probs": [0.2287050187587738, 0.07309851795434952, 0.35318902134895325, 0.34500738978385925], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "95d240915637f76cdf17dc6c21c52d877f9f33df527229cfc129fad5afc703a2:action", "state_id": "6665f9cff6fdcba7101cbab208a582a87a46b67d9ffcbd50cd58c3ff59b8c4a2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.66796875, 1.203125, 0.09375, -0.5283203125], "student_probs": [0.036226075142621994, 0.6396191716194153, 0.21092401444911957, 0.11323073506355286], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8dffc7c77531b780733d0a6b013e4a2906f31f4c213fe3e1c44605fdc4d86d13:action", "state_id": "301956fc5a570b0e45279ed231e422abe6495648578616480228d302c850cd02", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8740234375, 0.71875, 0.46875, 0.42578125], "student_probs": [0.07454010099172592, 0.3665410876274109, 0.28546249866485596, 0.27345630526542664], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "29e02e5117b1d34e8bd67823bc5c8ccf2b946777588e1ccad236b971f01958c6:action", "state_id": "1d7701d3446620f677533b0f2db44118be93449c3309ecd991871e7faf359744", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.78466796875, 1.501953125, 0.39453125, 0.56640625], "student_probs": [0.05569489300251007, 0.5481283664703369, 0.18110692501068115, 0.21506980061531067], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "542d68821cf5723693b7371be5fedff352ec8af9e0661ac8feae48a4a82c8a8e:action", "state_id": "5a52c7029bf99e181fa0f0180500cff753ebac0f7f5f0660249671955d2e28fe", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.86083984375, 1.943359375, 0.296875, 0.29296875], "student_probs": [0.04189930856227875, 0.6919187903404236, 0.1333509236574173, 0.13283105194568634], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5ec39a2f67a6e52bda17c6b59bd18e9c07526a42a43bea1738b6aab8df83bee1:action", "state_id": "9e0f992c2f6335d565aa70dd282df024ba94201e0be3c45fa32343e37b61eed1", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8837890625, 2.025390625, 0.125, 0.1328125], "student_probs": [0.04024498909711838, 0.738163411617279, 0.1103629618883133, 0.11122855544090271], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e74026175d8df6a00e22efca954129e498c180d1e36bb8d12542cb7c89259b4a:action", "state_id": "b76adca25313b7a77cf1e3db6486803928b582c4f55afd7b4402d191d4a5c536", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.66015625, 2.162109375, 0.0625, 0.3046875], "student_probs": [0.04444606974720955, 0.7473563551902771, 0.09155434370040894, 0.1166432648897171], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f2815b2d52be23b7e434d913678ea17e47fc3b8ea60dc00d060819ec00370e51:action", "state_id": "c9d48d4f24e5bc32fbbe0f073aefc4d115e6bf926211709076f47625d170f5ce", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.412109375, 1.15234375, 0.44140625, 0.6015625], "student_probs": [0.09188096225261688, 0.43919652700424194, 0.2157260775566101, 0.25319644808769226], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7cabbd32d1120c9aa44b5a1242e312ff9e2fd177b104767c2d9be84d1e303d11:action", "state_id": "280f40f1d196eb4b651398957a986347242a36a3a5ad118a14f6b1143719d5a8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.234375, 1.38671875, 0.55078125, 0.869140625], "student_probs": [0.08876173943281174, 0.4490119218826294, 0.19463226199150085, 0.2675941288471222], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b77ad7a40b69de613c494bfcf8607209becf553b4a42498f6df11afedfa82a8b:action", "state_id": "057a4e5a574db062328485296d20ff022269a4b2bd1f97d6c0ed4b4bf245a0f3", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.57421875, -2.9921875, -0.8818359375, -0.8740234375], "student_probs": [0.1903013437986374, 0.04609202966094017, 0.38031187653541565, 0.3832947015762329], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5e27a5aca40ca9b8e8074e59d55c53c0d31c5f3a6217370355bf3b5b315ff2e4:action", "state_id": "09b246455ed5667ce1f04747dc29a673a35ddf2508a5ff79ff66165951620188", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58203125, -0.591552734375, 0.38671875, -0.79150390625], "student_probs": [0.07657671719789505, 0.20618455111980438, 0.5484209060668945, 0.16881786286830902], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2fd488955824ad8e50f2ee322c94933f0651cd53cd96cf47e752b0c606fcb4af:action", "state_id": "d3721271eb6a7fbb476938316d4c505157eddf0cb004a912550d5267706e7451", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.2890625, 0.27734375, 0.6328125, 0.515625], "student_probs": [0.2149217426776886, 0.2124178111553192, 0.3030882179737091, 0.26957225799560547], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7e91c0869f799e933562975cc2941d5157428bec4bcb260a906b1886b1ad4b5a:action", "state_id": "e216dbc2b5616b55b895330c07e936605155b4c34c1f62dbbae7511a261e428a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.15625, 0.98046875, -1.8984375, -0.26171875], "student_probs": [0.08068514615297318, 0.6835318803787231, 0.03841188922524452, 0.19737111032009125], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0faa30d009d6430802db409343a3c55ba39340b6190812ee483acf6874fd112:action", "state_id": "ddf836d10ab7bda0bf469ec57bd4ffafd89aa7c5ac4b12c63b97b3d0a046f2ac", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.796875, -2.2265625, -2.3203125, -1.220703125], "student_probs": [0.24860736727714539, 0.16177190840244293, 0.1472949981689453, 0.44232580065727234], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "71ed6f0a1c175057693c1d60b73f085b71d4101c3e4615b123d315ac92560a2f:action", "state_id": "9f2008fa950a837d10f81f1bacce08734bea99e706dd6fdbe3f426bef845cfb7", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8740234375, -1.55078125, 0.15625, -0.8173828125], "student_probs": [0.18627627193927765, 0.09467719495296478, 0.5219148397445679, 0.19713161885738373], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9bdcf2038348b7814f8704fe3255c752d386b5b5ad01ecea7cd149780bcd08fc:action", "state_id": "de17a4e880d47e2c7feab109dc95488e92ae96fcacb591db28117d05e0ab62d4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8203125, 0.3671875, -1.7421875, -0.8681640625], "student_probs": [0.07360825687646866, 0.6560632586479187, 0.07958950102329254, 0.19073893129825592], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "35cae20c4e63eff73c86b805b353a7adb3c2d38d5156869e0a69eee780fca7ae:action", "state_id": "a821dee9c173de96b7dc4dc25f20e1ee9a773538f67b72ee4621926858e493a9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1328125, 0.33984375, -2.6640625, -1.2890625], "student_probs": [0.06342428922653198, 0.7518246173858643, 0.0372852124273777, 0.14746586978435516], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2ef5fdc2f772cb55c504e1f961251c61844b0378264bcf204c71ec790bef2d73:action", "state_id": "4584b4962639b012cd9f21a9bce6b6fd88ee72d6674860ff7ad5202ae023ef12", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.76171875, 0.42578125, -2.4609375, -1.171875], "student_probs": [0.08187605440616608, 0.729753315448761, 0.04069022089242935, 0.1476803719997406], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9376e378344abfb69f1e66699006942a746824ad4500d20528e288c4fc4dd94b:action", "state_id": "168b74056a329256acbf9de3a9c9bcd29cd5be151562e8aecb4f3c50b94b133c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7578125, -2.76171875, -2.74609375, -1.7578125], "student_probs": [0.36514192819595337, 0.13380451500415802, 0.13591161370277405, 0.36514192819595337], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0316f2fa6152d498d1fa4eb5718755449f3827ad6b1d163cf186064606f7aad6:action", "state_id": "0b9d39c05359589ac4142073683b46da25ccc2faa96411e73beedc5d5352b3fb", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0009765625, -1.515625, 0.18359375, -0.9384765625], "student_probs": [0.16859179735183716, 0.10076911747455597, 0.5511740446090698, 0.17946501076221466], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2d7fe27ae52d3d2a423c163649a44d9de710c4f5fa800c81c4df3b397f7e371f:action", "state_id": "5eb889996c0225b8c5fac16f276e117c8ee58e0fb1994b4bb830b21ecddbe0df", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.515625, 0.31640625, -1.296875, -0.6201171875], "student_probs": [0.09141051024198532, 0.571001410484314, 0.11376222223043442, 0.2238258421421051], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6b290285e5948db20b8cef8f4427c1e0fe6c3305ff3216959a69661fe74ad070:action", "state_id": "c333d29716095199d1ab496f63d970cc2c069233e72bab9ab054d4da7ccf828c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.92578125, 0.17578125, -2.5625, -1.36328125], "student_probs": [0.08723704516887665, 0.7135065197944641, 0.04615061730146408, 0.15310578048229218], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c0863758e522a3f7bec8032a96000fb9facf9ea89d82b2299f673976080eb4de:action", "state_id": "55042703e5d98880371963a028bb88c890f21193c053bf62ae6fb61b711924e8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8359375, -1.98046875, -2.54296875, -1.56640625], "student_probs": [0.272636741399765, 0.23594743013381958, 0.13443878293037415, 0.35697704553604126], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0968edd9d2d9d01db54bbc7aec9474ca762e61183392dd8360e8b520e3f1c57a:action", "state_id": "4c32553e9f96769f1abc27ca565ba53e8e894394786917bbdb201fb9575c44db", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.96875, -0.9736328125, 0.603515625, -0.015625], "student_probs": [0.1063096821308136, 0.1057918593287468, 0.5121522545814514, 0.27574631571769714], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2c3bd00fecb698e721b6452a61c7d3ee56073f3ec36e5d5c1a74a5eb06736da1:action", "state_id": "12f64225ec95e66b529bd2847bea82de3498911a2153db85483a7caa16b41aad", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.134765625, -0.416015625, -1.03125, -0.61669921875], "student_probs": [0.17124143242835999, 0.3513646125793457, 0.18991754949092865, 0.28747642040252686], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "91e249312998ae76fc832dc463070e253b08fe23942b5b916642ce7adb6200e0:action", "state_id": "062b3b2f7d47da626f14432b5450872e2682c4ce48789d2c7db838050cb88604", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9921875, -0.23046875, -2.9296875, -1.31640625], "student_probs": [0.1089370995759964, 0.6342793107032776, 0.04266038164496422, 0.21412327885627747], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dffa072f4873682caad990bd51c64452619e1e670f2ef2df2306c952ffb14e32:action", "state_id": "de91b152ea424f0a287d98ad2b5a9f271980eba205360aa4daf6b5bd4889228b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.23046875, -0.505859375, -2.94921875, -1.8984375], "student_probs": [0.11776500940322876, 0.6607005000114441, 0.05739408731460571, 0.16414044797420502], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fb8666a6e3a6496f43908c7083b622ee6638229dbc66102c6bd99bdcd9439672:action", "state_id": "1b5affd2217c4474b1c4c868e53563b063487df3d2b6e08133db2aea6e6d1cb8", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.6171875, -2.90234375, -3.296875, -2.4140625], "student_probs": [0.2870348393917084, 0.2158205509185791, 0.14546217024326324, 0.3516824245452881], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "addc589455bfd1f0039e1c285ee977141767fa98ee7f5b2e1491b3348a9caf97:action", "state_id": "bd6e09c15f30ff13ab00c07576441006d91e302db4dda8c02940aad89f6fb8e2", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.01953125, 1.134765625, -0.263671875, -0.8984375], "student_probs": [0.030036181211471558, 0.7039445638656616, 0.1738620400428772, 0.0921572595834732], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2a098ad7f0f96c58ed7f5a19b182d599cbca310895a8378b4985f854b9e56c0c:action", "state_id": "3969634aecdb2cbc4f065c48adcf450c27049eb640bd6250e10a364fa5acc64b", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.099609375, 0.64453125, 0.296875, 0.38671875], "student_probs": [0.06586407124996185, 0.37680724263191223, 0.2661546468734741, 0.2911740839481354], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4a7244bbba5095a3616f1e6eaefef4fb7a290165aebc542d112ef921f8b29b9a:action", "state_id": "2905501e8b9bccbe994dfff81f1343e7943437ba602c496aa19a59214a95f49f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4140625, 0.748046875, 0.8125, 0.876953125], "student_probs": [0.0889471098780632, 0.2843344807624817, 0.30326420068740845, 0.32345420122146606], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4b8984f888bcdc405b61d7eb34efd32c1e1a5e0ec0150175ab1df60dfc1240c3:action", "state_id": "7e3df451e11f7fa7a46b662d7dd5c997b777408b16b562cfdd2f342a8cacd3d4", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4140625, -2.73828125, -0.00390625, -0.07421875], "student_probs": [0.2493988573551178, 0.02440614067018032, 0.3758573532104492, 0.35033756494522095], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f74b94d7d67af2e002e115c6d2dfa2d268b32d5962cab95045b8b6b39df3bc6a:action", "state_id": "8bb4cd9e11582a78334eb32d0fbde483c72905235c3beb22cdcbf8e8c905ec98", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7578125, 1.1640625, -0.07421875, -0.5849609375], "student_probs": [0.0354708656668663, 0.658909797668457, 0.1910061091184616, 0.11461322754621506], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "65beb23314a8055294a89a42069a1de142dab2bc0677ee3b0c530bf11ad67fdb:action", "state_id": "639e03bfea7450967b75f8102a30f0e11c0b2419aff271302bcaea8b6f06be38", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9287109375, 0.734375, 0.30859375, 0.40234375], "student_probs": [0.07403617352247238, 0.39058271050453186, 0.25515174865722656, 0.280229389667511], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b800dc4f4d188addb442a7a3637998c1561c7a477d361138a28361cdca66add7:action", "state_id": "e75a4f6d28273b18e20ab137afcd0c2ebb9d3539d501ebd41d4311f1d9457bba", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.66259765625, 1.576171875, 0.41796875, 0.630859375], "student_probs": [0.0589153878390789, 0.5527312159538269, 0.17358523607254028, 0.2147682011127472], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bd26588d4a5f23e5c7b6cf49213defa8957e566ee41a91684a1a4e93293b0eb7:action", "state_id": "95313e4c00b904a5c33c89739acc373fbeb4e35e9bd1fcf27834f3607723df06", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.71630859375, 2.005859375, 0.35546875, 0.4375], "student_probs": [0.04483484849333763, 0.6820846199989319, 0.13094313442707062, 0.14213742315769196], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "205673355482caacda0660843d5867d1bfc2a6eb993fa6eee16c7043f51213dc:action", "state_id": "d96b7c517394d93d76ea467c259cb8db8982008cee52fe519d0ec75f3b7ece0d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9208984375, 2.001953125, 0.1171875, 0.12109375], "student_probs": [0.03959941118955612, 0.7363207936286926, 0.11182110011577606, 0.11225874722003937], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f4efa71fac6e4e85e49160a3f46388ccab11267b09d562eacc8716aa712db1fb:action", "state_id": "c7442a80804a09101edc743151f1bef722a1e9f6acc68e60e4a86723e330cc18", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.037109375, -3.36328125, -0.28125, -0.50244140625], "student_probs": [0.2026756852865219, 0.019795116037130356, 0.4315858483314514, 0.3459433615207672], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a92ea3a4703b365e06ec48cdfcb179dfb35361f69bbe51a2a8c3ac166598d9ba:action", "state_id": "11cc543e7b9c4e12bf73b280aaffef901528581f80345875d97bdee39aea4f61", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.546875, -0.6519775390625, 0.375, -0.72509765625], "student_probs": [0.07964710891246796, 0.19490322470664978, 0.5442891716957092, 0.18116044998168945], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aa20457ca748b40e80f0e08afaed9c2e2f258064ef830a621e1fce37b82dffb2:action", "state_id": "06d5119e7851726d26692b0ea4cad53c59f902f09ba7cf2cef72b1d6667d0857", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.013671875, 0.26953125, 1.080078125, 1.263671875], "student_probs": [0.2612447142601013, 0.12412846088409424, 0.27918198704719543, 0.3354448676109314], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "65f7596322d31b853c13d26a712c32d0b7d4af19224adeded8ce1fa41ca96e45:action", "state_id": "d17b9fb0267cd931fc9d41e17c825197c42123bd6a4f7494480353cc9cd1fbc6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.267578125, 0.9140625, -1.201171875, -0.173828125], "student_probs": [0.07186519354581833, 0.6367853879928589, 0.07679951190948486, 0.21454983949661255], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8a2196fe182292245d86e6ffa08d1355cd9bd59c6ab8a82fa7b4daa7938c2a2f:action", "state_id": "adc937a56d430c0bdceddf9b1d0ba27b93d441f9b1ac2af49eab1c9740ffc80a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.82421875, 0.5625, -2.3203125, -1.1875], "student_probs": [0.06955593824386597, 0.7566116452217102, 0.04235292971134186, 0.1314793825149536], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "65b2c214feefaf936772bcb2a60fed54df303f0a35e3e523f57a78899cebd568:action", "state_id": "5f9b34ca8b8fb6744fb9a18ca304c031f81814edd43b37f18466b1de44c67790", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.37890625, -2.91796875, -2.86328125, -2.1953125], "student_probs": [0.29403942823410034, 0.1715116798877716, 0.18115243315696716, 0.3532964885234833], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "195e396608f7fedaffe2e937bc761249d892f01f903f58b6f6a46682e6bafcc1:action", "state_id": "1f47c5f138550b7beb5315a9a3496abcc807afe4a20d1aa4da26ec410faa40a5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.98046875, 1.03125, -2.07421875, -2.2265625], "student_probs": [0.016436003148555756, 0.9079533219337463, 0.04067949950695038, 0.034931205213069916], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "159483016898cc435864dfa58aea9e71ca6dd951ca38dafb4d56e17419a4759e:action", "state_id": "2bdbf3dda39d03ed219aec74d873d881afef9446620f40d30d95aebdc840ea1a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.06640625, 0.65625, -2.140625, -1.30859375], "student_probs": [0.0518597736954689, 0.789341926574707, 0.04815016686916351, 0.11064820736646652], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "90ee8fbb70560048cc42021f24d05e63f1cc75e8b405e46c5f19c43e114fb001:action", "state_id": "362ac017aed765b0fff6a51f868cc3e5058d15c22ca56a057cde3ecbf6d0580f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9140625, 0.46484375, -2.5546875, -1.23828125], "student_probs": [0.07000045478343964, 0.755521297454834, 0.03688764572143555, 0.13759064674377441], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "758e22db4988c60e52a806afb39a4e915fb5dbb24094d0b40fbb3488c05e5e0a:action", "state_id": "61897e8bb92a2483ba5a495ddb8f97fc660577e711ce6fcd76a96e6c7d23ed0e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.51953125, -2.88671875, -3.421875, -2.20703125], "student_probs": [0.2885890305042267, 0.19989976286888123, 0.11705686897039413, 0.3944544196128845], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bf3e73f56a53993c1c51a9f3bba52adbbeed02bcdbfa4d15b46f148e63bfb90e:action", "state_id": "7cb5a07edce0a8fb66048985ccefedab50bd82e0514b94c9a32af90b47c4132d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.50390625, -0.6636962890625, 0.34765625, -0.7119140625], "student_probs": [0.08407311886548996, 0.19478508830070496, 0.5355259776115417, 0.18561582267284393], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fdc5685b5674ed63fcdfeebd830c65a8319e19ed6b74f93f8f849026214b8648:action", "state_id": "8e1b6f961af3b40548da8a0fa5139183249c82853db5e145e63d51bd8ef10c8f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3359375, -0.01953125, 0.974609375, 0.841796875], "student_probs": [0.10721103101968765, 0.1471136510372162, 0.3975600600242615, 0.3481152057647705], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8bdfaea5141ff316a5b86775839351d59e8f72c0e2483669898f9451671a095b:action", "state_id": "fd25bfc7aece5f3970cf6c58c7e741e2d66a10473367b9136fa08e5116999d26", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3681640625, -2.546875, 0.234375, 0.1015625], "student_probs": [0.22028881311416626, 0.02493390440940857, 0.4024128317832947, 0.3523644506931305], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1dd57dcc776c9ab37d2870e274f4ad150b40ef238785d1516e24327ae8039c58:action", "state_id": "38daf6fc4e5964c62f66ef3621710ce29487e7ecfaf6248cec8bb478f150c843", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0390625, 1.087890625, -0.4228515625, -0.94140625], "student_probs": [0.03141146898269653, 0.7163194417953491, 0.15812471508979797, 0.09414435923099518], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "35badaa677d170b6773fde1eca17190b1691346166f8e4f268aa11fb9059450f:action", "state_id": "b8e4ebb030d22e280e95ef9d4ea96083bbeab4078307259b2db41cb1583cb192", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.044921875, 0.65234375, 0.23828125, 0.49609375], "student_probs": [0.06785868108272552, 0.3704405426979065, 0.24484625458717346, 0.3168545663356781], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "125a0d58ceb21653f9dafd01b58944ae35ab80c63124d376520be1317599dfbd:action", "state_id": "0d8c1db8f28be206522f9cad9a854efc3ffe5259283219f19aad3845cf56588a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8212890625, 1.416015625, 0.40625, 0.61328125], "student_probs": [0.05562134459614754, 0.5210633873939514, 0.18982566893100739, 0.23348954319953918], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7d987e9d2393ff0dfbc7ddf4bfe32eb639d517a7a87a7ea85ab57b7184a6ed1c:action", "state_id": "e3d877600b0752f412efdfe50186e1e3e8bade5baad22d2c7c55ff06e4de06a5", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.982421875, 1.79296875, 0.2421875, 0.3125], "student_probs": [0.04149646311998367, 0.6658062934875488, 0.14120568335056305, 0.15149156749248505], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bda3bbd4762ec4cf792e8a20caec343c3a0361049f899b4ce94c3ca2169c3b0a:action", "state_id": "1e3e739b8bc596d06c17c59254932de4bea3b8fba8a13318d66d3ae0149d8c59", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1796875, -3.6796875, -0.435546875, -0.5694580078125], "student_probs": [0.19890321791172028, 0.016326971352100372, 0.4186180830001831, 0.36615175008773804], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7ec524f1c49d4fff33e0ff6cf231e18e694857e3edafd4033ece3d45a522accf:action", "state_id": "4ea8661819c7c648215479fee8e47a16ce160a058ff406945f96a4a674daff04", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.109375, 0.8984375, -2.2109375, -2.28515625], "student_probs": [0.01645759306848049, 0.9056015610694885, 0.04041594639420509, 0.037524934858083725], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "924e4f79d8a99a197b8c7577a8883fcee162d527b56086acc229b654af0ca434:action", "state_id": "a30006c5bb06f52a8fb8b01e5e71a4135a0fa9417dfb89f4f269c155c5d28483", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.20703125, 0.61328125, -2.12890625, -1.130859375], "student_probs": [0.0458783321082592, 0.7699344754219055, 0.04960630461573601, 0.1345808058977127], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ef9f4a75eeef47dcc629f9ac841c9c550e132034da7f3b707e4284f42a650d86:action", "state_id": "fc76709c9f9759bd1190f43a4543d60e76271264a2e54ab75daa9eaa2d6b4398", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.89453125, 1.43359375, -1.7578125, -0.921875], "student_probs": [0.030602125450968742, 0.8533710241317749, 0.03508550673723221, 0.08094141632318497], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "558420939051becc4726fc7a033bc703bc37ce20c5679cd1264f343e1a75cbb1:action", "state_id": "caf6f17a183a0db281b0f78851d43092fdc7c4bd28d1b8417ffb40529750778a", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9375, 0.8515625, -2.21875, -1.0703125], "student_probs": [0.04901765659451485, 0.7973096370697021, 0.037000469863414764, 0.11667218059301376], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c883b3b4f8f1e97674b7367a32327e41a93a698a1833e09323e5e135e4b40f31:action", "state_id": "9f2ef0123f22b3b83cd1dc8b00c540197bb63b6eb4eafb3bc1a46718296f4bb6", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37890625, 1.439453125, -1.42578125, -0.4443359375], "student_probs": [0.047059543430805206, 0.7882167100906372, 0.04490453004837036, 0.11981921643018723], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5c4b009a9fe57291df86c68eddcc51e08ba62bf528b0e68defa8442bf7970784:action", "state_id": "3158e03028738d36d3999e69e92cb9398d643b2c2771819f7b1f7199cb4a91f9", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.609375, 1.599609375, -1.75390625, -0.62689208984375], "student_probs": [0.03414083272218704, 0.845119833946228, 0.029546426609158516, 0.09119289368391037], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "54ebb568bda1b1e7a66a5b664f87e23b9b2d53a9c9349947cdc09572a9b78c57:action", "state_id": "4e2bae84af7dad9ca39016834736bbbfef929a42a3281cc522f0b2867ebaf726", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.046875, 1.595703125, -1.1953125, -0.5716552734375], "student_probs": [0.021783893927931786, 0.8319306373596191, 0.05104631930589676, 0.09523911029100418], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "36f243a85f297d20ac738f5ffb3c1449819898ccbad25f03b7004d734bac738c:action", "state_id": "075e9f7a8d2cb0e4ae7d922fbbab4fca93cc629ae7afcb2af1c930ba50de136d", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.80078125, 1.79296875, -0.9453125, -0.3974609375], "student_probs": [0.02283554896712303, 0.8305336833000183, 0.053720101714134216, 0.09291069954633713], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f626e310c471e7aa1b3b8266bb76f32cf0f93ac4478e5bbb10aa621c7205de51:action", "state_id": "40fa6a9ca6f8acabf4b7bb46ab84713c245c82f10a023af6ed61b1dcfd00b713", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.52734375, 1.71875, -1.166015625, -0.775390625], "student_probs": [0.03306204080581665, 0.849357008934021, 0.04745177552103996, 0.07012917846441269], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a70dce3cf6ec98bca3fe08394a549f8249e0d90bba6004570abb8204e5ddb93b:action", "state_id": "864ff21c2170534a61c6ee7246a5edf41efbe234d1b5383c8dc31ce627173f24", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.91796875, 1.66015625, -1.65234375, -1.578125], "student_probs": [0.025306642055511475, 0.9061383008956909, 0.03300608694553375, 0.03554895520210266], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "be4da511401d5c615c3ded4fda976f8c50c1d26a14780e8f4dfb0266c5cf2da9:action", "state_id": "b2d85f7cac71bd028b3c8f97646a93aad19cf684c4c309c4c4d5b5df7e1334ff", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.97265625, 1.771484375, -1.703125, -1.3359375], "student_probs": [0.021518204361200333, 0.9096317887306213, 0.0281748715788126, 0.04067517817020416], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8d20f808b3abf3b2f653fde71f51b493ab4a474f423132854b970df280a898fb:action", "state_id": "20157002d0e7a45986d478395c7dce3428d01397b19436928ab2a490e88388ec", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.05078125, 1.375, -1.80859375, -1.27734375], "student_probs": [0.02841886878013611, 0.873785138130188, 0.036206576973199844, 0.06158945709466934], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d3ce872af36cb32c85e6dfc4349ac3bfc6a1bc7a0c33317e1e608de32ec99651:action", "state_id": "a5836e74c4f3988ca2ffbf0a9ee9e4788bd9ebd427eebee373f201702c1f407c", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.80078125, 0.76171875, -2.3515625, -1.54296875], "student_probs": [0.06313613057136536, 0.8187617659568787, 0.03639793023467064, 0.08170421421527863], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ffaf3016d46b515dc26e82cd6bf80215bf458bfdad44fc25be3b0616e8de4476:action", "state_id": "f6d19889b9f824743416647d35d76de73ae246b5a3b46aca6410004f58a66049", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2109375, 0.90234375, -2.453125, -1.85546875], "student_probs": [0.03890068456530571, 0.875060498714447, 0.03053349442780018, 0.055505409836769104], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f5393f2d4c3fc5a5af1194c460928ab245367815e44ebb12334e865edc04fb53:action", "state_id": "a370cfc24ef8d2b64a8e7a8ffdbc9a0446a7ab60ded5cc21a4d5874f5de3b74f", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8125, 0.77734375, -2.47265625, -1.3515625], "student_probs": [0.06086420267820358, 0.8111791610717773, 0.03145282715559006, 0.09650382399559021], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2b67f23736ae76013ffdf7ca73eed23445b74123c7dc547c816eca4432d78778:action", "state_id": "6fbb038e59007bc628f36d30dea4c28362d9ee35aa85d4c7557e74d32e3e8ebe", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.953125, 0.6640625, -2.48046875, -1.51953125], "student_probs": [0.059417326003313065, 0.8138477802276611, 0.035066355019807816, 0.09166856855154037], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0675ffabe1845233e7d75397888480f9cc13def313bc2bc495b46f4327b9acb:action", "state_id": "850364134fbb9c1b96a6015312a5f29ef0903766b6bda49b56629f307bdbe40e", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.224609375, 0.609375, -2.02734375, -0.982421875], "student_probs": [0.11134729534387589, 0.6968976259231567, 0.04989495128393173, 0.14186014235019684], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b3e6683d905eb6f8ccb7e4301b922945fd3cac584826a80f034349133dfd8ca1:action", "state_id": "bd6a0444a2c27eced42e6ebe765cc1d5ac0ac6fa50f65b2c6e8d605563e047fc", "family_id": "unified_shooting", "split": "ood", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.203125, -3.25, -2.9921875, -2.03515625], "student_probs": [0.3346492648124695, 0.11747294664382935, 0.15202128887176514, 0.39585649967193604], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"}