{"id": "7675c640eaa1bcdf20d122b9e9773d80a7d98009767300ff4e305d8d5e59e50b:action", "state_id": "c15fc2e39ce68a7ff7763e5009b0e913cbfb834b0f62246d626226e5756585b3", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.04, 0.22, 0.67], "teacher_probs": [0.07, 0.04, 0.22, 0.67], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.6796875, -1.2890625, 0.91796875, 0.9453125], "student_probs": [0.0336533822119236, 0.049736469984054565, 0.45203956961631775, 0.4645705819129944], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.04, 0.22, 0.67], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "73fcb745f0156d36f95be4b6e038d3e951ba24d0b38f2c5e15653d835fdffdc6:action", "state_id": "3fc193943a0b05556150ad20505d8f76e00acb0e8d6d18662f171be6b8652f6d", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.12, 0.52, 0.36], "teacher_probs": [0.12, 0.52, 0.36], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.126953125, 0.2265625, -0.5791015625], "student_probs": [0.15150265395641327, 0.5864683985710144, 0.26202887296676636], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.52, 0.36], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "90a50a53be7feb063fb97a89cdd173e2b75c669ac0b76c72a81aadd271b0152d:action", "state_id": "5e40bea098f3e1478460956193a896efabd390db06b0f19f172a60a6c7e5fc8e", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.19, 0.36, 0.45], "teacher_probs": [0.19, 0.36, 0.45], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.302734375, 0.1796875, -0.16796875], "student_probs": [0.11745303869247437, 0.5172159075737, 0.36533111333847046], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.19, 0.36, 0.45], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "307baa076ef97e63a0ff39ef3888c8d49b4e9f63cb41a96d29128b24882b5436:action", "state_id": "5bc7c0c8cbcb357b80e29334c23774d63482e65d9cd17a055796d007576922fb", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.16, 0.6, 0.24], "teacher_probs": [0.16, 0.6, 0.24], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.94921875, -0.388671875, -1.5234375], "student_probs": [0.1371326446533203, 0.6529467105865479, 0.2099207192659378], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.16, 0.6, 0.24], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "af4816b73d42669eb8c3d38c4f0954206a9ee8c9de88de4c4b36b16f16ed2400:action", "state_id": "70e64bf5578bc46280f6c9901707bfe3764a4df553b81ce9162e63fc1a52a8e7", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.31, 0.69], "teacher_probs": [0.31, 0.69], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.0625, -1.61328125], "student_probs": [0.38954654335975647, 0.6104534864425659], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.31, 0.69], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "57dbe1c126e89a4513d4a55acf2347d6d7b7b3c786fb075f848da88da60ecacc:action", "state_id": "1ac4254cc6e960bf7fcd724d618cb5ee1d1b862e926e98ab63a168013c41eae7", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.12, 0.76, 0.12], "teacher_probs": [0.12, 0.76, 0.12], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.66796875, -0.52978515625, -2.140625], "student_probs": [0.21077311038970947, 0.6578426957130432, 0.1313842236995697], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.76, 0.12], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ebcbf0a7ffd650474acd579bbcd07004282bd36681d1832889b84241efd3d3f3:action", "state_id": "40962fa89d0f85937db1ce837d7ce215e84d3afffc1dd07d8029e92d6d6740ce", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.58, 0.36, 0.06], "teacher_probs": [0.58, 0.36, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.09765625, -0.6279296875, -2.296875], "student_probs": [0.5884659290313721, 0.3462792634963989, 0.06525484472513199], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.58, 0.36, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "73a9b7f9ddfe2e2b0e73855539e7db08530795ad0b3a23e6c3b245a32effd0bd:action", "state_id": "eeb0b8706f1047c4c36219603dbbdc15814e5e010757937083f71a513ec5772c", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.71, 0.29], "teacher_probs": [0.71, 0.29], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.677734375, -2.29296875], "student_probs": [0.8341368436813354, 0.16586315631866455], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.71, 0.29], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6ea8b278085750c4f1d74afa05225e804e81ac67b5b610da6b0c89adc9ed2a9a:action", "state_id": "51cf0e917d65c40360775bfc002e5c47e4cdb0aaf152ebcd508aaa3326343542", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.26, 0.67, 0.07], "teacher_probs": [0.26, 0.67, 0.07], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.470703125, -0.455078125, -2.09765625], "student_probs": [0.45202335715293884, 0.45914170145988464, 0.08883500099182129], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.26, 0.67, 0.07], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fa2ed11507b972fb85f1cbdfe2e92d2e13a7c75b2697b430affead3830618630:action", "state_id": "700cf1d83329f2157208e468d9b74b920d9044b19550c6204eee2bc67e855b02", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.82, 0.11, 0.07], "teacher_probs": [0.82, 0.11, 0.07], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.123046875, -1.67578125, -2.1640625], "student_probs": [0.7453980445861816, 0.15777722001075745, 0.0968247577548027], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.82, 0.11, 0.07], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "690562a1b80308397bf923f308b0143d5ae465607ebd07e6165f183ad87b4cb4:action", "state_id": "9b5d920d3b21bf5d1c1d3746508966abc10c67709ce4457ca2a844b696781958", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.39, 0.61], "teacher_probs": [0.39, 0.61], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.78125, -1.48046875], "student_probs": [0.4253665506839752, 0.5746335387229919], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.39, 0.61], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "21a826a9424e19820846e652eb2e0290af6bbcd8b58a166c76a834ccfd5cf389:action", "state_id": "a85616fc5e4f386e2aaa20fc9c3ae66d294bb201f5023ea048e47a51fc65fd42", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.91, 0.05, 0.04], "teacher_probs": [0.91, 0.05, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.16796875, -2.20703125, -2.4296875], "student_probs": [0.8565586805343628, 0.07967237383127213, 0.06376896053552628], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.91, 0.05, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4a5e050d3f3dea3f9e092d21dc4bb268993021a1d46d3ace70eed4b31656f5cd:action", "state_id": "c915bf47daf56c8135c7c3dd6f89ced9193a54b27f4d12f6aa00f7bfc4f7b540", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.57, 0.39, 0.04], "teacher_probs": [0.57, 0.39, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.46875, -0.095703125, -2.0390625], "student_probs": [0.6060175895690918, 0.34462466835975647, 0.049357835203409195], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.57, 0.39, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5a58e2ec66e5beb141113892b5449575dad5e603ddeb15f44713ae1feb1fa4db:action", "state_id": "641b85da39ca96109d6ac7fe0cddd877d7c8734e23d6e8c97f8c814047240147", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.77, 0.23], "teacher_probs": [0.77, 0.23], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.4453125, -1.94140625], "student_probs": [0.915808916091919, 0.08419107645750046], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.77, 0.23], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e8834db780a0bb82d30f9f13d80e8c3e660a4dcd2cca5e4c5c8f36a3c413f8c7:action", "state_id": "a20e107eca453c677355f966cffedbf1cc4919f60672b1bff35f424ef2f33735", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.62, 0.35, 0.03], "teacher_probs": [0.62, 0.35, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.11328125, 0.109375, -2.01953125], "student_probs": [0.4729015827178955, 0.4710579216480255, 0.056040506809949875], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.62, 0.35, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b8ec9f568252f79db15034e898a162221f6774277fb965c23f75f7fbfcc8ff97:action", "state_id": "2bb7744ea729f91b659a88d3e60f58df081b25c1aa0b9367345879cf26f13a48", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.72, 0.28], "teacher_probs": [0.72, 0.28], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.09375, -1.796875], "student_probs": [0.8688268065452576, 0.13117322325706482], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.72, 0.28], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "16e45c4c159ff4036e7d7e6d3cdf8f4d402144ccf7eb10113693d07a30f8d94d:action", "state_id": "d84439561547b7a869523a00d3bfc89dbdec5b8c2ef89757042bd496b592297a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.35, 0.27, 0.38], "teacher_probs": [0.35, 0.27, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.0703125, -0.0859375, 0.0546875], "student_probs": [0.32075491547584534, 0.3157820999622345, 0.3634629547595978], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.35, 0.27, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "81fb4fef5f5221ea14eabace10b23107270b4d6b554fe8a8cfa8816e4c944b85:action", "state_id": "9d7e5ce8487176a7cca0a68764b79daf8255f60f88b02f75db152812364e69e2", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.59, 0.29, 0.12], "teacher_probs": [0.59, 0.29, 0.12], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.39453125, -0.30859375, -0.810791015625], "student_probs": [0.36373594403266907, 0.39637696743011475, 0.23988710343837738], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.59, 0.29, 0.12], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f629f86a9579bc87f3f284e4be92b50b55b8f26216ea09ede2707bb9c47d7c79:action", "state_id": "8b09189aa5b659b82471e62f33667196cf05ca7cfed8b11d30d5876c3f59eb39", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.69, 0.31], "teacher_probs": [0.69, 0.31], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.333984375, -0.521484375], "student_probs": [0.5467381477355957, 0.4532618224620819], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.69, 0.31], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b829a0684053a68d0cab380077c5cb5eb897bced937ec03106a62d06e3faba36:action", "state_id": "91ac6e37df91b673936e024070d25cdfc5f60d3fd5f4170e11bce6edba87cc23", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.29, 0.31, 0.4], "teacher_probs": [0.29, 0.31, 0.4], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.09765625, -0.24609375, 0.1953125], "student_probs": [0.35565799474716187, 0.25219929218292236, 0.39214271306991577], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.29, 0.31, 0.4], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "51f54f09a595869881e531585131eb1c0a685c8c0b937ed2bb46d04b4f365e9d:action", "state_id": "0a4aa10fe970e3f3684bf78c35dae90e348b07e8f1c324aac270a163fcc53c2a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.41, 0.59], "teacher_probs": [0.41, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.02734375, -0.1953125], "student_probs": [0.5554351806640625, 0.44456472992897034], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3175ef5ff7e0d14090ccd944c69ebc8401628d0a3250158f4106abffdc380c28:action", "state_id": "7d03a9c98af5becb8daf0fa7f06693ad9ae13f71cb7725250eae32fb37bbd5d8", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.15, 0.3, 0.55], "teacher_probs": [0.15, 0.3, 0.55], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.693115234375, 0.21484375, 0.830078125], "student_probs": [0.12397556006908417, 0.3073672950267792, 0.5686572194099426], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.15, 0.3, 0.55], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2f7002fef8d074b47726dc6de80594473716157509a687789486a48f68141cd1:action", "state_id": "a9de808bb87958e89e8b84090b745fe62c24e60217b897bae7c45fa0c8ec0ec6", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.25, 0.44, 0.25, 0.06], "teacher_probs": [0.25, 0.44, 0.25, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.38671875, 0.703125, -0.03125, -1.0546875], "student_probs": [0.16910722851753235, 0.5028926730155945, 0.24129053950309753, 0.08670956641435623], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.25, 0.44, 0.25, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "65e82ea8cc79762609858fccb5f902e86bd8e8199dff7b343bc8ce8c31eac755:action", "state_id": "30efaa40a90a662d77eca9125d13ade0cf42f2fe8389d193aa7a58550642ac51", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.49, 0.42, 0.09], "teacher_probs": [0.49, 0.42, 0.09], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.0234375, -0.51416015625, -1.90234375], "student_probs": [0.5780642032623291, 0.33767613768577576, 0.08425969630479813], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.42, 0.09], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9e98c46927230d8672d21c34be0aa812cc9e39e82d49590d0612497adf5bd86a:action", "state_id": "942b4f129a90304305cfa2c4498a237e0ce8dc5c0675c6328458633d5fcf5aa8", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.6, 0.4], "teacher_probs": [0.6, 0.4], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.741455078125, -1.9140625], "student_probs": [0.7636159658432007, 0.23638400435447693], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.6, 0.4], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1e92f398184afe69d9bee00a32ce4cd130f740b0d53938c7e5258e319ef7a2b5:action", "state_id": "c739d02cdf1514de417bca15b20da7b94caf6cc5227b2367b83fa8cd8a221f7e", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.66, 0.25, 0.09], "teacher_probs": [0.66, 0.25, 0.09], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.03125, -2.11328125, -2.54296875], "student_probs": [0.8293212056159973, 0.10339703410863876, 0.06728173047304153], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.66, 0.25, 0.09], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "10a9f737a8bf33986cf164ba5aee4340044b1c9b2f25768a892e7f2041b14f27:action", "state_id": "b6f9aaff372bea3d3adbdc26d012005faa13fa4af2552ba08dfc15ea5b824832", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.58, 0.09, 0.33], "teacher_probs": [0.58, 0.09, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.310546875, -2.25390625, -0.375], "student_probs": [0.48058393597602844, 0.06883019208908081, 0.45058590173721313], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.58, 0.09, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f7d80c3d550abe50ce4f87e01137400a4d59f7cedbcd808b29127fad0dd78e59:action", "state_id": "9e26123c4252c55e0252e76ce107d420cd1a136e1b4dd0ce20ef28519f4f1bec", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.39, 0.11, 0.5], "teacher_probs": [0.39, 0.11, 0.5], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.3466796875, -2.140625, -0.14453125], "student_probs": [0.4183518588542938, 0.0695730671286583, 0.5120750665664673], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.39, 0.11, 0.5], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "600da8a75a8bc4aa01bec814e1d7eeca558e1435043bf5acb5e25be6e690c29a:action", "state_id": "74f5d9ceda23d7e44a91b9333c7118735d849949120953708eb8f3b7d071e873", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.84, 0.16], "teacher_probs": [0.84, 0.16], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.396484375, -1.96875], "student_probs": [0.8281063437461853, 0.1718936562538147], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.84, 0.16], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "187300f2674ddb59e04726d4dbffbdae184d7479838134806d07ea4a40150604:action", "state_id": "c5d423dbad3100b8c75a8ad1eae9a95092cadc28c26ea64ae222987b6fef1403", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.25, 0.09, 0.66], "teacher_probs": [0.25, 0.09, 0.66], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.24609375, -2.04296875, 0.0390625], "student_probs": [0.19739563763141632, 0.08897317945957184, 0.7136311531066895], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.25, 0.09, 0.66], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2f920b07d117b57e05f69d91414dafb7d074976aab506e81dae04610cff1b109:action", "state_id": "63d12064ddf30a6b547b31d95dd5317c3f36ffc3fd4c64899027168451a56aa9", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.66, 0.34], "teacher_probs": [0.66, 0.34], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.197265625, -1.7890625], "student_probs": [0.6437773108482361, 0.3562226891517639], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.66, 0.34], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4ec7dcbc005ee27d4f2c12926ba525b3a2b931b88dd2a74ceb228404d0d1e6ea:action", "state_id": "3678fa65c5362ded581bd35c40d6ad9bc22c52820be7e5f79397926afec96fbe", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.1, 0.12, 0.78], "teacher_probs": [0.1, 0.12, 0.78], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.83984375, -1.65625, -0.169921875], "student_probs": [0.13309766352176666, 0.15992049872875214, 0.7069818377494812], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.1, 0.12, 0.78], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "30f59caa3a4125373593df5c958339771b0f366ddb780f287f9ff3db68be9282:action", "state_id": "c2f3b53ec108c173fae4440a63d38004036098a5d5fd1bdf07cbd1be12aafa9c", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.08, 0.34, 0.58], "teacher_probs": [0.08, 0.34, 0.58], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.68359375, -0.658203125, 0.078125], "student_probs": [0.10405155271291733, 0.2901149392127991, 0.6058335304260254], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.08, 0.34, 0.58], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e54bbf394cdcc1f3ed75ab5b2f4c22de9b8ba9ffa03e9a5d3f2e4069dc79a1b6:action", "state_id": "aeee7b8165c96e007266e557637e495e028c86e65164d91e088605ed8e6f127f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.26, 0.74], "teacher_probs": [0.26, 0.74], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.89453125, -0.943359375], "student_probs": [0.2786492109298706, 0.7213507890701294], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.26, 0.74], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b326789ca762d0134a485c00d08b9d5eb6acc26dd4084d873def1fd89b429445:action", "state_id": "4e2a208310da887d50b0da4798b59cbe732e42b4a1f828fc56dd756f0a939b77", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.38, 0.58], "teacher_probs": [0.04, 0.38, 0.58], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.71875, -0.310546875, 0.01171875], "student_probs": [0.09318013489246368, 0.38097649812698364, 0.5258433222770691], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.38, 0.58], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "09c705e4087cc4cbf6c88e9020492901c70ee95cbab78201ff2c3515b23ff85a:action", "state_id": "598bc0b3275c222804c8016144f2aac1ddd250b34a1a5792a94726b707d67691", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.18, 0.82], "teacher_probs": [0.18, 0.82], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.68359375, -0.4130859375], "student_probs": [0.21917034685611725, 0.780829668045044], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.18, 0.82], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "158349161fea47ba388226326b7521801057b6efb7b64e3a3993faa2a933867f:action", "state_id": "a282817e7cb0a1505a36a7d997418bedfd195798a5661cb63f083254b04d78bf", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.09, 0.84], "teacher_probs": [0.07, 0.09, 0.84], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.453125, -1.3046875, 0.6484375], "student_probs": [0.09672152996063232, 0.11219893395900726, 0.7910795211791992], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.09, 0.84], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "171bff4d199f5e55ea2873616194c886d3f7f6453c956ed36c90db5225dfd9b2:action", "state_id": "3a7d5b452fc2985144eccfeaf515f0142634d8d2891249f762040b14161af0ed", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.49, 0.51], "teacher_probs": [0.49, 0.51], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.5234375, -1.37890625], "student_probs": [0.4639299511909485, 0.5360700488090515], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.51], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a469341d0d13533face8800276403f5802496d090d27ce8dc74cb42c05f0275d:action", "state_id": "06705db861a4d3002f6d2c6e64e32c26d22bd733fd7963d115ef9514cc3680e1", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.06, 0.87], "teacher_probs": [0.07, 0.06, 0.87], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.20703125, -2.52734375, 0.44921875], "student_probs": [0.06262250244617462, 0.04545906186103821, 0.891918420791626], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.06, 0.87], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "44883fbf4e3477b07c22a89fc65052a1aed78c7302b693322208fd612f756a6f:action", "state_id": "e5de40ad001cbf79012deaac0ebb16556d87301d1f3a096ceb32318c8e5ff6be", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.39, 0.02, 0.59], "teacher_probs": [0.39, 0.02, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.0625, -2.625, -0.2890625], "student_probs": [0.5644491314888, 0.03841124475002289, 0.3971395790576935], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.39, 0.02, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d4d03a772c343810d9690a20cdfe029d2df7ac0baf166423daff75361b9ffffa:action", "state_id": "9bd7e683e0131bbbf2a9abb523c4171525a9ee4960a6050db977ef82a76ade11", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.41, 0.03, 0.56], "teacher_probs": [0.41, 0.03, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.173828125, -2.3671875, 0.02734375], "student_probs": [0.42837995290756226, 0.047782108187675476, 0.5238379836082458], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.03, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6edb57c0b0e850395c6819dcae1f0b4bcd681b59a7dd634023a3ac2d6625d738:action", "state_id": "444563468293151c9f54de8a92e4c08e62865caa73ece5f5dfc2285d4175c649", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.1, 0.01, 0.89], "teacher_probs": [0.1, 0.01, 0.89], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.185546875, -2.44921875, 2.66015625], "student_probs": [0.05459222570061684, 0.005675846245139837, 0.9397318959236145], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.1, 0.01, 0.89], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6b9b8996b388cb5abfeafe0376f4f286dffe91bddfa0721c71deaa017e79d0fb:action", "state_id": "e3ad23dcb7e95aff35c8d789dc21f6e0acc6acd302a04fbf5db1b7a245387906", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.96], "teacher_probs": [0.04, 0.96], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.33203125, 2.4765625], "student_probs": [0.008093290030956268, 0.9919067621231079], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.96], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "74e5ec2fa29a4dfa7fff934fd96fc6e52a61f3ea8d67ff6d1e40a0c94d04a100:action", "state_id": "f42ac6bc1b291397b4479b6ff086056556652fdf819dc285a8bce53be2f406cb", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.37, 0.51, 0.08, 0.04], "teacher_probs": [0.37, 0.51, 0.08, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.314453125, 0.3046875, -2.046875, -1.32421875], "student_probs": [0.29424822330474854, 0.5465164184570312, 0.05203944072127342, 0.1071959063410759], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.37, 0.51, 0.08, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c1ad8dac1c22023ab9536b7ed92ca6a7da37f6a598b5b5f9699a19dd0988033a:action", "state_id": "b183c3c1e7e8d8e777a0637a5775f78fec20a0e4c67f16bf210b544f0b1e4d84", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.65, 0.29, 0.06], "teacher_probs": [0.65, 0.29, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.291015625, 0.02734375, -1.12109375], "student_probs": [0.35575973987579346, 0.48912349343299866, 0.15511666238307953], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.65, 0.29, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e32c7db2382f4a08028c96df6f58a41f4c2c4424c6f14001e61a007594726dae:action", "state_id": "7c377e5b5b534c4294cd59cbe702ebb215e41c2b6ac222e9561d128fc5965695", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.74, 0.26], "teacher_probs": [0.74, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.664306640625, -1.208984375], "student_probs": [0.6328999400138855, 0.3671000897884369], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.74, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "75d37225c52d32164ba97767515d0d3969e401c821c7be0ace2b11a11ddf4383:action", "state_id": "4df90b1b7dc9bb144434501a6229fedbac1f7901ffbcb96c68a487f29bc313e9", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.49, 0.45, 0.06], "teacher_probs": [0.49, 0.45, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8564453125, -0.79296875, -2.11328125], "student_probs": [0.4255160689353943, 0.453402042388916, 0.12108185142278671], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.45, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "99364f5bdc714bf8223619ac257d302f50f344b8b358a06196d027dbcc97b266:action", "state_id": "48ecdad2adc2734296a0d1655992596980dad58868292ab00ac4027462740e94", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.3, 0.29, 0.41], "teacher_probs": [0.3, 0.29, 0.41], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.19921875, -0.73779296875, -0.515625], "student_probs": [0.43246176838874817, 0.25237593054771423, 0.3151622712612152], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.3, 0.29, 0.41], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e8109de07b5160a2c1d6d5110f5e5a04fa2dd4cd43a5d65ea2e6cadce921a90e:action", "state_id": "c412a5e85deb087a81950bc65ff63b43ac512921fb19833e7fcbcb3118d60ee9", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.5, 0.5], "teacher_probs": [0.5, 0.5], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.0390625, -0.7623291015625], "student_probs": [0.6733258962631226, 0.32667404413223267], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.5, 0.5], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9b12025c2d3f07d34163631a6717e3e1babd11d0f6c199da915b48a5ef0de95d:action", "state_id": "de07e7e669de36c0aa0b04d7754954db2c643a5fb5fb5664d37f8968662a84f6", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.24, 0.34, 0.42], "teacher_probs": [0.24, 0.34, 0.42], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.30078125, 0.2578125, 0.3359375], "student_probs": [0.21558783948421478, 0.3768933415412903, 0.4075188636779785], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.24, 0.34, 0.42], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ee400df386717caa084a7974cc8e318180bf3624f584e0fc1d9634252b883639:action", "state_id": "e62ddacbc9aa1f14f998f4efb48eafe6d1964781365c10302f508fa43241edb4", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.33, 0.67], "teacher_probs": [0.33, 0.67], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.171875, 0.47265625], "student_probs": [0.3442229628562927, 0.6557770371437073], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.33, 0.67], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "700308cdef1970233d77da58ec74d1aed49579c76be13bb3ef5c0e875bd84531:action", "state_id": "bcabfa55a100bd5994fa84832fdec1bea7431d66c37953c3a848c1df37a1237f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.29, 0.33, 0.38], "teacher_probs": [0.29, 0.33, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8544921875, 0.6875, -0.12109375], "student_probs": [0.12893183529376984, 0.6026134490966797, 0.2684547007083893], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.29, 0.33, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "818d017ed559d2e155f0798aba0f655ad78ec3c6fb0303c8624486dee8dc5982:action", "state_id": "9576f459758637ad76496d2e371c93f7114433a48a53c7a120226a4089a0efdd", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.3, 0.7], "teacher_probs": [0.3, 0.7], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.5478515625, 0.57421875], "student_probs": [0.2456274777650833, 0.7543725967407227], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.3, 0.7], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5887d40e1f59c96bb43d8ee801fb93b697254989ae59546e95f4c0ffa8c2be99:action", "state_id": "a95e0bde4b7106808fbdc67202f1ffb063d0efd8d7e8767dcda8ea7cf3d580eb", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.5599999999999999, 0.14, 0.3], "teacher_probs": [0.5599999999999999, 0.14, 0.3], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.78466796875, -1.359375, -0.173828125], "student_probs": [0.29369890689849854, 0.1653142273426056, 0.5409868359565735], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.5599999999999999, 0.14, 0.3], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "69c5952c048c3309dcbb6a3dd51fbbbb318c6d46ee2eb20a10649fe31c190f06:action", "state_id": "02db99841c7199b368b8508af6069474b8f72ac52c2c6dbfe64ebe78f87ee0c3", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.24, 0.12, 0.64], "teacher_probs": [0.24, 0.12, 0.64], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.224609375, -1.84375, -0.2265625], "student_probs": [0.23521745204925537, 0.12664271891117096, 0.6381397843360901], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.24, 0.12, 0.64], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ba6083c51cdbc3346b675072e778360bccd656a153d0fa5380e2b493ef891ed0:action", "state_id": "c92c06ce5650b93673a1cbf211340b1e4024a81728565b72e05d0457811b504f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.69, 0.31], "teacher_probs": [0.69, 0.31], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.44921875, -1.6875], "student_probs": [0.5592900514602661, 0.4407099485397339], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.69, 0.31], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a3d33f861580e145e6dab6f043befa713e28cb5a88140d3d98d75622bf44128f:action", "state_id": "23a6758af3dead2894d784aa2932aeb019edf0176aa76422f76aa07ddbbe5310", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.15, 0.14, 0.71], "teacher_probs": [0.15, 0.14, 0.71], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.62890625, -1.58984375, -0.330078125], "student_probs": [0.17528991401195526, 0.18227267265319824, 0.6424373984336853], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.15, 0.14, 0.71], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "629e0c2527a6b30a6c3a13b3ce21fa8b986ccb4511bc41aa1b0f48fc8accefe1:action", "state_id": "9480a56a8a379281d40fc6e60ad824161877dab73ff58320328e0886d3dc339c", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.12, 0.4, 0.48], "teacher_probs": [0.12, 0.4, 0.48], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.51953125, -0.203125, 0.04296875], "student_probs": [0.1052551195025444, 0.39260080456733704, 0.5021440982818604], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.4, 0.48], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4ed621564a37af2d5d6fdb4b2e67f4ec8e1eed7ba9df7480631412ad328b3020:action", "state_id": "7e684a8d1dc5269463ebb0fff97ba2a7699bdd97788fa1fe62032e4ab8c3bbbc", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.12, 0.27, 0.61], "teacher_probs": [0.12, 0.27, 0.61], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.93359375, 0.6015625, 0.40234375], "student_probs": [0.10586928576231003, 0.49145057797431946, 0.4026801586151123], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.27, 0.61], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "468d9d8d5f9fe798bbc043d6ddf741306c9edaab8a6358ff3a93bbae8c89d1e0:action", "state_id": "3a65119487295ca4fb847c7150c968563564c2c54c443a1e374ccf3e9c1a3e53", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.18, 0.82], "teacher_probs": [0.18, 0.82], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.9638671875, 0.49609375], "student_probs": [0.18847329914569855, 0.8115267157554626], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.18, 0.82], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6360576da46e8b438ebf17addd9a20055f0ed295e2ff976f1e2ad810ab9b659a:action", "state_id": "dd2d6e2663140195b5c5cade92e1c95c57a069befed681335972057f0d262588", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.1, 0.84], "teacher_probs": [0.06, 0.1, 0.84], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.9560546875, -0.322265625, 1.00390625], "student_probs": [0.10016237944364548, 0.18877990543842316, 0.711057722568512], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.1, 0.84], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b839f192398975b17915a1bdd86c5c86e7b534f1c49686982a2c8628520d3d31:action", "state_id": "6d5162d98e128c0abed277d9676c94957194dbf494ab77607a3c440d74f27d65", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.35, 0.65], "teacher_probs": [0.35, 0.65], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.166015625, -0.4658203125], "student_probs": [0.3317689299583435, 0.6682310700416565], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.35, 0.65], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f381a7f8af8bcc05b4001466a172e912302ace3b69849e607b12d441e220a2d2:action", "state_id": "4b96f013ba01ed5379b0c9d501905f6091fbcf9b7db650a72c432aec3672cdb1", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.03, 0.93], "teacher_probs": [0.04, 0.03, 0.93], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.62109375, -2.15234375, 0.91796875], "student_probs": [0.07014758139848709, 0.04123763367533684, 0.8886147737503052], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.03, 0.93], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0a70c890b9bfe2c711d224eabeea15a13b2478769f48b0903aac825079f8804e:action", "state_id": "b5e4fd318da8f3c69553ec8f84fdc9f73d7c41173f4c47e6223a13df5a7f7d5d", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.01, 0.94], "teacher_probs": [0.05, 0.01, 0.94], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.28515625, -1.94921875, 2.583984375], "student_probs": [0.0903378278017044, 0.00967147946357727, 0.8999906778335571], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.01, 0.94], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fd2e4b60dd57c4b9c4324bbea46e5b087f84f01826c614afb085e76ac20e8902:action", "state_id": "fbdd720ab445dcea5b1cb84ea3c724690429919f256a33fde27705029429563c", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.22, 0.37, 0.2, 0.21], "teacher_probs": [0.22, 0.37, 0.2, 0.21], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.09375, 0.970703125, 0.697265625, 0.32421875], "student_probs": [0.13116884231567383, 0.3802916705608368, 0.28931066393852234, 0.19922883808612823], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.22, 0.37, 0.2, 0.21], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "22646f3a75d89edd48da5bad8b7cbe5456b82ba5eadcda3c2e603805db77ba7c:action", "state_id": "59fa5f3a2f1e02feae03891f95f0a6aa9d358e62e620868ef46fe1511e1e529e", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.29, 0.24, 0.47], "teacher_probs": [0.29, 0.24, 0.47], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.747314453125, -0.138671875, -0.5185546875], "student_probs": [0.24420173466205597, 0.4488269090652466, 0.30697137117385864], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.29, 0.24, 0.47], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fdd8fa923792555afc6fea2489171c0adcb3350154e81b21f5d41d8c52c082f7:action", "state_id": "f5618dadf3446ead3f622ff2c51e55e965ff965569cf2d17aa2deb218130f4be", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.48, 0.22, 0.3], "teacher_probs": [0.48, 0.22, 0.3], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.63623046875, -0.777099609375, -0.359375], "student_probs": [0.3137177526950836, 0.27249619364738464, 0.4137861132621765], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.48, 0.22, 0.3], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "57805cf4e4285c65c8a237c61308dc1bfa1d0e34ac5ea848443aebf1b78ade4a:action", "state_id": "dab26c02c227735705b7833da7e6230e068fdb0e8adf45cee01ab01748ff1927", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.57, 0.43], "teacher_probs": [0.57, 0.43], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.5791015625, -0.296875], "student_probs": [0.42990800738334656, 0.5700920224189758], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.57, 0.43], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d70acc46055074c86aea8580225546ec01e8d87d591d92f0b614c987a9a38865:action", "state_id": "2b80c2484375b9aa47a67f3a0dc14f4976fcadda5479b807b9b197fe6c305368", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.7, 0.16, 0.14], "teacher_probs": [0.7, 0.16, 0.14], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.5078125, -0.142578125, -0.966796875], "student_probs": [0.5711968541145325, 0.2980744540691376, 0.13072875142097473], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.7, 0.16, 0.14], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "02ae75a7a2f8e1cfaae36e9638173661ad1ff2ce69561be498ee91ab396f2dfe:action", "state_id": "61daa48926ebd077b40279b210b454c8e95977b668d2c9663d4b945bd8b9998b", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.41, 0.49, 0.1], "teacher_probs": [0.41, 0.49, 0.1], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.140625, -0.4638671875, -1.47265625], "student_probs": [0.5728739500045776, 0.3129906952381134, 0.11413528770208359], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.49, 0.1], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "06b3d5d41ea481662017e32084251fea8b979009fbdba35f4814e0476f30e74c:action", "state_id": "471b34c2c8725567a517da1934e375b241658597b1f47d48274b897663ef5a58", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.88, 0.09, 0.03], "teacher_probs": [0.88, 0.09, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.39453125, -1.3671875, -2.04296875], "student_probs": [0.7942002415657043, 0.136403426527977, 0.06939644366502762], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.88, 0.09, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "12c1fc0dd5b4f15c291fb45fe8cb254386fe0e3e025152ac597883b2d0f0569b:action", "state_id": "1e1ffae4ad1b6681aa4817356255b918fa6fe1b4e3dbd35382779ee6109d6872", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.58, 0.1, 0.32], "teacher_probs": [0.58, 0.1, 0.32], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.03515625, -1.28125, 0.15625], "student_probs": [0.4172181189060211, 0.11185494065284729, 0.4709269106388092], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.58, 0.1, 0.32], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4299fec751dd65cb6af3b5f8e857239b86ee9ad93c20c562574c45868da5ae28:action", "state_id": "d761fcd5516b8e883514a7491d49f377fadba50f353573bc07d1859c3d952973", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.57, 0.1, 0.33], "teacher_probs": [0.57, 0.1, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.287109375, -1.208984375, 0.1171875], "student_probs": [0.3453013598918915, 0.137351393699646, 0.5173472762107849], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.57, 0.1, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "83b20d54c2b86c50df24a1bebaefd247cdf4e46f90351715e2b60873aabfdc4e:action", "state_id": "6334f40fc170ff96aa414adf5189bbd5d6daa89dbb8920ee5cce39d1375b45c1", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.62, 0.38], "teacher_probs": [0.62, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.37890625, -1.69140625], "student_probs": [0.5774953365325928, 0.42250463366508484], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.62, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "dc84d038e56941743ebba7ce87ada40e9ab048184d1ce1991b67413916f0253b:action", "state_id": "cf46ae7e393b46c1f897a9af84f00c5a32c23e12cd5e03da663682bbe112ebc5", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.15, 0.21, 0.64], "teacher_probs": [0.15, 0.21, 0.64], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.11328125, -1.82421875, -0.53173828125], "student_probs": [0.1389346718788147, 0.18550212681293488, 0.6755632758140564], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.15, 0.21, 0.64], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8b1b7e9abb351150d6254b90a6ce499b78e258572d6a4894d5bac589e4f86ffc:action", "state_id": "59c411281930f7c3a47fcb34e68e703085645d742c9713af3fa16e76cc0bfb39", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.11, 0.3, 0.59], "teacher_probs": [0.11, 0.3, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.80859375, -0.51416015625, 0.15625], "student_probs": [0.08487000316381454, 0.3096845746040344, 0.6054454445838928], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.11, 0.3, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e07e71edc006c611ae071e4cff1703b9202ec445e1383fd5a5c393eb4fa721a9:action", "state_id": "fe8df469f8f190c0bae58f6be22a30e56761738823f8873af71e250ef5fb7459", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.29, 0.71], "teacher_probs": [0.29, 0.71], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.83203125, -0.73681640625], "student_probs": [0.25063756108283997, 0.7493624091148376], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.29, 0.71], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "edf761c95f4fff5641d0d74b31af27bfc66123a8cd6540902b263e2f4fe6442f:action", "state_id": "e1bf332b09dda782efae8e15962e2b41bd4844fab48532d17be2ec9a03e349cf", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.46, 0.48], "teacher_probs": [0.06, 0.46, 0.48], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.8359375, -0.107421875, 0.0546875], "student_probs": [0.07543870806694031, 0.42489245533943176, 0.49966880679130554], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.46, 0.48], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "763efd424db205efc25ad2b6f9e55e5e266afd45786c4cb90a6d8d3af5f355a8:action", "state_id": "05fe78c34ffacd97d8fff19db03605a3230f0c3eff5162f09809352e7bafc117", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.17, 0.83], "teacher_probs": [0.17, 0.83], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.8046875, -0.263671875], "student_probs": [0.17638766765594482, 0.8236122727394104], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.17, 0.83], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e6ae2f2a388a84ea6f16b7d053af7da05a5f347d679e756e698891ec722e7e1d:action", "state_id": "7dd88d54cbd1e61a77b515255955b5c923c066091c09ff1f0463552a926d535c", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.09, 0.84], "teacher_probs": [0.07, 0.09, 0.84], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.42578125, -1.47265625, 0.703125], "student_probs": [0.09652626514434814, 0.09210600703954697, 0.8113677501678467], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.09, 0.84], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e88bdaa1e7f7470ebf8c600cd434a45350b154e7e1a89fb0584bc8d7a40d9c90:action", "state_id": "3af47ca7e376fe3a0bf4f311d4f6c4708e76aea6c6e4a6b716843b41311e1984", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.46, 0.54], "teacher_probs": [0.46, 0.54], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.5625, -1.59375], "student_probs": [0.5078118443489075, 0.49218809604644775], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.46, 0.54], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "682f952309a523a2a55bfa2a9ebfa67419d9f04a41ec44daf6ac0ca48fa835e5:action", "state_id": "95b9a0f2b50eea9fa44da95e33faebc0e855f3c5f430daf265464750e68cf905", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.05, 0.89], "teacher_probs": [0.06, 0.05, 0.89], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.125, -2.671875, 0.23828125], "student_probs": [0.08193688094615936, 0.047421425580978394, 0.8706416487693787], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.05, 0.89], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d1789c82bd8102584b97ad466b793b4a7b7a5ff933a05e23af7e20a5eb0ac94f:action", "state_id": "f0a4aeae6cc22e4fc27eb3b9f5f452fbfedfb880bf574abedeb9ca433e3386af", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.52, 0.03, 0.45], "teacher_probs": [0.52, 0.03, 0.45], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.140625, -2.703125, -0.3447265625], "student_probs": [0.5974829792976379, 0.0347776785492897, 0.3677392899990082], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.52, 0.03, 0.45], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "428979a6bf8d945870785041317ddb79947d58e55e2e4ef64508d0fcfaeb7648:action", "state_id": "da9d609372e72ae9dc4766120b92bf15a36cd65d45fb7f9ca9792e1d572ab36f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.18, 0.82], "teacher_probs": [0.18, 0.82], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.24609375, -0.51904296875], "student_probs": [0.15096519887447357, 0.8490347862243652], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.18, 0.82], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c4f187b53995d7619092bff91a29a8e8d3d7aefdb85db9d3770f6bffd572d4bb:action", "state_id": "4691cef86fdb63aca6d26c4196f626882f39221557a369566e5ba789c834c764", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.4, 0.03, 0.57], "teacher_probs": [0.4, 0.03, 0.57], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.05078125, -2.5, -0.03515625], "student_probs": [0.47571274638175964, 0.041083045303821564, 0.4832041561603546], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.4, 0.03, 0.57], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4856d809da9c10a6ab30f665638c01d529769201a2cab7f2b9883130b0ec3bd2:action", "state_id": "6b73b50116bec2a42f1894a69840161e676436b6a11efd1a12fc5e07e60fa32a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.0, 0.95], "teacher_probs": [0.05, 0.0, 0.95], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.173828125, -2.55859375, 2.59765625], "student_probs": [0.05856703594326973, 0.005394643172621727, 0.9360383152961731], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.0, 0.95], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e233c68a5a9ef6fd7f2bd7ebe49384cdb50948ec893eb934343ea98b9d0dc8f3:action", "state_id": "a1d06a571c89fef259821442134813fe6061e7409b68a70eda8c3b528ba07d9f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.36, 0.03, 0.57], "teacher_probs": [0.04, 0.36, 0.03, 0.57], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.49609375, 0.609375, -1.6875, 0.37109375], "student_probs": [0.060581009835004807, 0.49742770195007324, 0.050027620047330856, 0.3919636011123657], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.36, 0.03, 0.57], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4ce483378c5751431f342d4d9c6fc0b8b74b0fb6e2a584a941959dc74dc35a13:action", "state_id": "9ac22992e16b634733a960b2677ce01642e719363ab676e5b712bcb392c8c71d", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.63, 0.1, 0.27], "teacher_probs": [0.63, 0.1, 0.27], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.08203125, -1.4921875, -0.644287109375], "student_probs": [0.5512596964836121, 0.1345653235912323, 0.314175009727478], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.63, 0.1, 0.27], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "158e682f5252346c86a79772aecbe37490daa1b5e45e237ffad015648406180f:action", "state_id": "b59b9808f1f1088c38ddddf7b530316cb150a10b4a73289518154ed6038031ca", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.49, 0.51], "teacher_probs": [0.49, 0.51], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.515625, -1.025390625], "student_probs": [0.37983837723731995, 0.6201616525650024], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.51], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cb025a0558dcef132429f386455d4bf9a47995f30489385d9e397978fcfab1c9:action", "state_id": "6cab4a26d0f03a2dde223e617c3e3dd45b4f3148c194dfb17c6477f9b3c436ac", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.41, 0.2, 0.39], "teacher_probs": [0.41, 0.2, 0.39], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.212890625, -1.46875, -0.739013671875], "student_probs": [0.5331279635429382, 0.1518513411283493, 0.31502071022987366], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.2, 0.39], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "47084532a1da077ae39a575792ca741b811f9ba7f18e4adc556eda6d6c932e1b:action", "state_id": "5cd5e2825d686bc459d6e18de830a684dad4c9efc2f5b9cc44a9519841fd0e11", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.24, 0.52, 0.24], "teacher_probs": [0.24, 0.52, 0.24], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.10546875, -0.3125, -0.5400390625], "student_probs": [0.40641534328460693, 0.3304133415222168, 0.26317134499549866], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.24, 0.52, 0.24], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6e3e188471d2546d5529ffa03f77a8196d33a5a7d43146798ac0d00bf868ce3c:action", "state_id": "54324d7d2a791d84397380fb456395fb1c125ab6025ff9101d67541389a84cc8", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.15, 0.59, 0.26], "teacher_probs": [0.15, 0.59, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.18359375, 0.09375, -0.1640625], "student_probs": [0.2994600832462311, 0.39517349004745483, 0.3053664267063141], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.15, 0.59, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f1eb63dc0c58ee03c23cc34e1a5204147285d761d4ae3004df4758f870e36248:action", "state_id": "783b82fd1c1cfa806241157192e43f9cc025495b04e58410730f0f0b3f509fad", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "west"], "gold_index": null, "teacher_raw_probs": [0.34, 0.66], "teacher_probs": [0.34, 0.66], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.14453125, -0.177734375], "student_probs": [0.5083000063896179, 0.4916999638080597], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.34, 0.66], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8fd814c480e9da4dc8c005bd74dac0d52db2fa04a1296757ff74571595cd90db:action", "state_id": "74c231da4d5aaf6b87f59bb08203cbf1241694cd7a2d383df173a46d593221f1", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.5, 0.28, 0.22], "teacher_probs": [0.5, 0.28, 0.22], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.234375, -0.291015625, -0.8050537109375], "student_probs": [0.3983944058418274, 0.37645626068115234, 0.22514930367469788], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.5, 0.28, 0.22], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "31600796839cdbd5a22721cfb888492dfd9aa13fb53e9cb82ac78bc1d4100486:action", "state_id": "86b02bbaad340b426b7bce60097212c4f8d9b49b58cdb6bfdfdf05b0ed7d172a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.67, 0.33], "teacher_probs": [0.67, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.42578125, -0.83544921875], "student_probs": [0.6010082960128784, 0.39899176359176636], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.67, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7aaa4bf2f3f3718c35f6a5875abf8ea90d78bef204f23f681b108b23de5cbfa7:action", "state_id": "0acafc11547ae0c0d867b7f0f782c8ecebc1c7dbcab27f38fb6ccdb8f35c5c21", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.43, 0.4, 0.17], "teacher_probs": [0.43, 0.4, 0.17], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.7109375, -0.78759765625, -1.8515625], "student_probs": [0.4452708959579468, 0.4124119281768799, 0.14231713116168976], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.43, 0.4, 0.17], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b76237e1f1f776b312f694c94c75b6b71ab7f1b18d3ab2f1283724d68a6b24e9:action", "state_id": "6582d48822477f1684724974ee627865422cfa526ef9e86939652dd58470e24c", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.57, 0.31, 0.12], "teacher_probs": [0.57, 0.31, 0.12], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.12890625, -0.74755859375, -1.8828125], "student_probs": [0.6451569199562073, 0.26854774355888367, 0.08629526942968369], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.57, 0.31, 0.12], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b71517da9f5b060d1aaf776043a8bc6ef1509eab8d9e030d41939d5e808db0ff:action", "state_id": "f168f8aa7e70b42da23733186bf9132d5f4c1e4f31269045c0bb6c0fb51cd865", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.65, 0.35], "teacher_probs": [0.65, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.78857421875, -1.8828125], "student_probs": [0.7491790056228638, 0.2508210241794586], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.65, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "faec8887cb1544872b85807e417efb54a96f889e70610224c24dbab99f814c94:action", "state_id": "753998b54934713d9e5d7809f1ca8f4a80ddd3ad982ec63a1bd65a9749a86f50", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.65, 0.18, 0.17], "teacher_probs": [0.65, 0.18, 0.17], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.1640625, -1.26953125, -2.56640625], "student_probs": [0.7670834064483643, 0.18291138112545013, 0.05000519007444382], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.65, 0.18, 0.17], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b77b61e9bf6766b8a237045611bd1c6e12fdbbbebd316c0e0a65959b9b45f907:action", "state_id": "9e3e72c255d04cf3dacec0909fa0852617aa0155a75c3a609264d8517e80d50c", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.55, 0.1, 0.35], "teacher_probs": [0.55, 0.1, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.056640625, -1.2890625, 0.32421875], "student_probs": [0.6343072056770325, 0.06075384095311165, 0.304938942193985], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.55, 0.1, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "70b1551f65ad0ef73d4bcd11f5e50637e70e5be8646a9e31ccdaeccf96ab1de8:action", "state_id": "4dd5459943a450e035bec42eec21fb598fbcee53e280b305079fe9fc0d567416", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.38, 0.17, 0.45], "teacher_probs": [0.38, 0.17, 0.45], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.029296875, -0.5673828125, 0.755859375], "student_probs": [0.5093392729759216, 0.10317583382129669, 0.3874848783016205], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.38, 0.17, 0.45], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5394d548791667ead1a272492a4e2e176f0c922e5056fd72fb4795a110c99826:action", "state_id": "2a9c75d3b1ba1f3081fbc10071a1981880cc95447d8930bcb9918c11a009f3e2", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.75, 0.25], "teacher_probs": [0.75, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.873046875, -0.580078125], "student_probs": [0.8104788661003113, 0.18952107429504395], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.75, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "274b4933658eea5bb96bc122301d2e91472a58dbb2337cb2aba355e58da05b90:action", "state_id": "e25cf931ff28291be3f5502affbfe13933450211061b70595350ec65cff66504", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.33, 0.08, 0.59], "teacher_probs": [0.33, 0.08, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4970703125, -1.26953125, 0.48828125], "student_probs": [0.24150921404361725, 0.1115470752120018, 0.6469436883926392], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.33, 0.08, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "98574c40866c3fd3ab3da9d12083471167af17508ec4aa7cd6cc2af04aff047a:action", "state_id": "95ef2a6876a03203c39aedee234662e7e17ad0904445124e9a35a920b923c47f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.78, 0.22], "teacher_probs": [0.78, 0.22], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.40625, -1.119140625], "student_probs": [0.6710395812988281, 0.3289604187011719], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.78, 0.22], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4851345aa4cb60d3a52c411e27e5b5b6d3bf7dcb554dfd1c955e81f53a72f56c:action", "state_id": "9e3138afb7d9cedec3e83a0f632fa7ee10e76bfa525b44356cece5453c4f97d7", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.09, 0.09, 0.82], "teacher_probs": [0.09, 0.09, 0.82], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.28125, -1.0390625, 0.578125], "student_probs": [0.11502507328987122, 0.14654575288295746, 0.7384291291236877], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.09, 0.09, 0.82], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fc60a8958c5c197e09836e9cb0b13d093271bb16703409d5f692de6bbbc53175:action", "state_id": "5ba7d6af5114af084dc0dfe2a7adeef12a04411f2bd710d75489e436328133a9", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.95, 0.02], "teacher_probs": [0.03, 0.95, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.859375, 3.4970703125, -0.07421875], "student_probs": [0.012319489382207394, 0.9606669545173645, 0.02701355330646038], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.03, 0.95, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bc9d0752a117c3e61ba7618b3e5f27ff23c8844a43b30bcf75efa592a6aebd84:action", "state_id": "4669b33fcc312767b786414f1fe5b7977bc036d494bb6c421cb7324eca2c9201", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.23, 0.06, 0.65], "teacher_probs": [0.06, 0.23, 0.06, 0.65], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.61328125, 0.52734375, -1.80078125, 0.27734375], "student_probs": [0.05897169187664986, 0.5015395879745483, 0.04888924956321716, 0.39059942960739136], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.23, 0.06, 0.65], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4a1ee033e85c6c6c826d8ae5e76bd03c5a70dc1c25367ef72eb47a4e5265c1b6:action", "state_id": "daa92c4ef96b55c96f215827917adacb68ed0d6a6cb41c26c1599466dfca2b70", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.42, 0.19, 0.39], "teacher_probs": [0.42, 0.19, 0.39], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.12890625, -1.306640625, -0.435546875], "student_probs": [0.5535087585449219, 0.13172687590122223, 0.3147644102573395], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.42, 0.19, 0.39], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "01a2d98ac4c874548a0764dc436719493e3d0f885900352f37645b546f01c33b:action", "state_id": "c2232e71ad7b7fb0ec1e839a5f6a12b7d6c97087d791642200eadff2c08c819f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.44, 0.56], "teacher_probs": [0.44, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.39453125, -0.81005859375], "student_probs": [0.3579041063785553, 0.6420959234237671], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.44, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a4d6cad45e2938cad46af059502edc54b11cc1edec2a245bbe9fec02ab8c25c7:action", "state_id": "8415d11f0132fb100683e88e1f64f8e23bdb5097cba9a64085fd69dc0cc9082a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.34, 0.18, 0.48], "teacher_probs": [0.34, 0.18, 0.48], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.0390625, -1.203125, -0.474609375], "student_probs": [0.5299286246299744, 0.15301787853240967, 0.3170534670352936], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.34, 0.18, 0.48], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "799eab8545fbe80d68b74b43c9de8b1980ced9f796a09e99c61d1dab6c4d340a:action", "state_id": "5e29fc70a93e90b10aba53a7f07a3de662f501a588f16b69fc1cf2e8067d8fd5", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.55, 0.11, 0.33], "teacher_probs": null, "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": null, "student_logits": [-1.12890625, -1.83984375, -1.4921875], "student_probs": [0.4573363661766052, 0.22463607788085938, 0.3180274963378906], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": "Rounded target has no identity probability-simplex representative", "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": null, "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "29fdcb6c85ca527ff2b7ba1bffb414ea183ae908c143a5cb47a21c90ae86fe24:action", "state_id": "d964d51bb17880ec0efbf35127585d2a5279e59ab7c4723fba4c6fbfcab59ad6", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.49, 0.51], "teacher_probs": [0.49, 0.51], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.60546875, -1.6328125], "student_probs": [0.5068355202674866, 0.4931644797325134], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.51], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f6cfd06c97570db0d84f2ec7fea1a8370a2fac21ddc8b507f410e6d32c4e766a:action", "state_id": "fa7b2b59a7bc4087843650a82930f1ae3c6f76dd30580b734f62e1e62b0df59d", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.5, 0.21, 0.29], "teacher_probs": [0.5, 0.21, 0.29], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.5927734375, -1.82421875, -1.2890625], "student_probs": [0.5585650205612183, 0.16302859783172607, 0.27840641140937805], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.5, 0.21, 0.29], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "952ba1731b24e31552faf4933964663d82143f25dc3fb0b415c08f314b497339:action", "state_id": "631c538a0d91ca0c16bd231b6c395458b0fd369c9a2fb756fd5c77d429a9f500", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.43, 0.51, 0.06], "teacher_probs": [0.43, 0.51, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.08984375, 0.04296875, -1.43359375], "student_probs": [0.4603695571422577, 0.4392876923084259, 0.1003427729010582], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.43, 0.51, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "46422a4bdd450aa230d3ab9b8874625442238ac8579847bb894e0db62ce659cf:action", "state_id": "c1add2e8d2b228b29fe281890a1ace14123fd4a43848f9e0b562017f353fc204", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.58, 0.39, 0.03], "teacher_probs": [0.58, 0.39, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.228515625, -0.185546875, -1.9140625], "student_probs": [0.448581725358963, 0.46827682852745056, 0.0831414982676506], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.58, 0.39, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bd80cf4abc3641c58d6fcdcf7a4f17038ff298ef4c212234f7a3be2c30506cc0:action", "state_id": "60fd37594cedcf7fdf0949875a09379f2e3974d8c0d4f63a7fad08a1e3c12e84", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.33, 0.44, 0.23], "teacher_probs": [0.33, 0.44, 0.23], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.1015625, -0.162109375, -0.181640625], "student_probs": [0.3491261899471283, 0.3286149203777313, 0.32225891947746277], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.33, 0.44, 0.23], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "74b98d67f6f0f21034909e8484c75092d74abd9ff788ebe965ec3a5612187ffc:action", "state_id": "762bd6439bd585fb084db0020461293309b2cbd78f066e86aa6a7698c7a0f8d5", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.47, 0.53], "teacher_probs": [0.47, 0.53], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.1640625, -0.43359375], "student_probs": [0.5669777989387512, 0.4330221712589264], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.47, 0.53], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4d1afe7610d87a194042624be1fa343016b2fb41943d518066bc16696a9c17c3:action", "state_id": "549f1b51ebee40a823799662e5da9cbb7fd413ec15d347c5e57c461fd4641990", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.36, 0.39, 0.25], "teacher_probs": [0.36, 0.39, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.1171875, 0.14453125, 0.0390625], "student_probs": [0.2883273959159851, 0.37458375096321106, 0.3370888829231262], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.36, 0.39, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "490c72b8ce2f735fdeae2fa727b5fd5a6ac20bbfd99d76ec26d7ce14635799fb:action", "state_id": "5870856638d048419d67c94ebe501fc880274507e64d77e17b57caa73222fe4e", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.41, 0.59], "teacher_probs": [0.41, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.138671875, 0.125], "student_probs": [0.4344612956047058, 0.5655387043952942], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b2dca1580573c0eb95d1b829db007636c77725ba5faee7237758837b3ff899d0:action", "state_id": "4ed6a9f0291dfc3d2ba2e2f5dc68f438b4adcb30f627edd0e34df78b0a66cba8", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.24, 0.35, 0.41], "teacher_probs": [0.24, 0.35, 0.41], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.404296875, 0.1640625, -0.228515625], "student_probs": [0.2526818811893463, 0.4460766017436981, 0.30124160647392273], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.24, 0.35, 0.41], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b47592a5acdb739dd4f083ebe17b21fbc581cb814d1c78b045258b85bff07df6:action", "state_id": "1295fb504ab624fb29aab2fa204cc2a85d8a8a1d00e2ffbbeb520de3f91a008a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.44, 0.56], "teacher_probs": [0.44, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.177734375, -0.05859375], "student_probs": [0.47025004029273987, 0.5297499895095825], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.44, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e38c85cb9e7b5d8dada6f81291ce5ba3b4ca672b3d740f999eaf61c524afad49:action", "state_id": "3d47a8e34377c2551708d3d1fa748335245c2d28aef56da48073b0e8e58d8c94", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.68, 0.13, 0.19], "teacher_probs": [0.68, 0.13, 0.19], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.19921875, -1.45703125, -1.4140625], "student_probs": [0.7193798422813416, 0.13729605078697205, 0.143324077129364], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.68, 0.13, 0.19], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fb64bec1e1e559e5a3fe531ba38b30dff4b5de311415a993df15d9603581637b:action", "state_id": "87d1e78367624ab6b40e8f0697f51066a3724f2ada348f188b1c60545203cc34", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.35, 0.13, 0.52], "teacher_probs": [0.35, 0.13, 0.52], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.623046875, -1.26953125, 0.38671875], "student_probs": [0.2342555969953537, 0.12272282689809799, 0.6430216431617737], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.35, 0.13, 0.52], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6a7bccd12be1cfdd1f47f17a3309ba032a42fba0e6489bb7d94bf2829b7519a0:action", "state_id": "17417e8b35043046b3c2ab20a2f6b162e746cceca350472c110518c469ddcb1f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.67, 0.33], "teacher_probs": [0.67, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.666015625, -1.05859375], "student_probs": [0.5969031453132629, 0.4030967950820923], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.67, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ea4f1abf0820eb833a87063ae6391244db8182754ecff12aa8b6c829006e3a57:action", "state_id": "7fb912c6a3d68e374f922a3510659d18543a5a347e94a0fabc5432c85b376b55", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.12, 0.14, 0.74], "teacher_probs": [0.12, 0.14, 0.74], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.59375, -1.212890625, -0.0546875], "student_probs": [0.1403752863407135, 0.20544511079788208, 0.6541796326637268], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.14, 0.74], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a63fdfbeddc045826558a45710a8b72642a10a49f322109cb279bba8edd179c4:action", "state_id": "17865b0b62b1420304c8e30ebd260e1f1b0fc74b5e931bc183c207eb70ed6828", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.94, 0.03], "teacher_probs": [0.03, 0.94, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.19921875, 2.921875, -0.33984375], "student_probs": [0.01538738701492548, 0.9482724666595459, 0.036340147256851196], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.03, 0.94, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3e591aa7d166c2057e18cfd726a1a35e2e7ad24a78d5bf313a10c5b949d399ad:action", "state_id": "9b606726eee2f5878cca9fe0032daac1cca43cf9ca7ce6f6a4e99c3044dec8ca", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.6, 0.38, 0.02], "teacher_probs": [0.6, 0.38, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.3671875, 1.2109375, -1.8203125], "student_probs": [0.5272536873817444, 0.4509839713573456, 0.021762358024716377], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.6, 0.38, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5923e2168ed6ac17296580f8ceb8a9b87c1c419ba1fd1f8acbe668be64dcf998:action", "state_id": "3f3decc8d6401dd4d35d0687bcd09b88ff4a9e8463b2f0b60e6589b83df69fa0", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.45, 0.51, 0.04], "teacher_probs": [0.45, 0.51, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.5, 1.087890625, -1.4140625], "student_probs": [0.3392513394355774, 0.6107158660888672, 0.05003279447555542], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.45, 0.51, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c482da6954de58c4fd85ec10fe457e0bfe830b7ccb97dad672baead37a388e2f:action", "state_id": "4e2a816b3cfd1f2e0e9b2f3a82d6c2eff2fedc00a52d7ba967bbb3c9026f45be", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.75, 0.2], "teacher_probs": [0.05, 0.75, 0.2], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.98828125, 1.31640625, -0.111328125], "student_probs": [0.028757451102137566, 0.7833537459373474, 0.1878887563943863], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.75, 0.2], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c3ece9b09212769299140b6531c6f01ac25ba20a1246b30abaa852a8563937fb:action", "state_id": "b694659be255639abc25e2bd3b80e8dc9fb5c08bd4199eccdba0fa24034f562d", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.64, 0.31], "teacher_probs": [0.05, 0.64, 0.31], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.40625, 1.43359375, 0.5625], "student_probs": [0.039565082639455795, 0.6770808696746826, 0.2833539843559265], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.64, 0.31], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "47f2e4f5e3f0d99784e04ec69488abf84b12fcc8c8fe66b4bfc8d5df078d9692:action", "state_id": "8dddd163b19bfb30aca9b4af3db505d9fb0b2b532ca2bf7a8e22b4ad223446c9", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.46, 0.51], "teacher_probs": [0.03, 0.46, 0.51], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.67578125, 1.13671875, 1.640625], "student_probs": [0.022117717191576958, 0.36829307675361633, 0.609589159488678], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.03, 0.46, 0.51], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "016977844c7d2cdd7eb516588ecc6ed737b37308c95d9b898c62fa03945e0e1e:action", "state_id": "cd333be2ce5cf324e407d3a7ae6d84ece7bb217027a9c63d618cbcfea4451de0", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.27, 0.02, 0.71], "teacher_probs": [0.27, 0.02, 0.71], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.73388671875, 0.5625, 2.9384765625], "student_probs": [0.427160382270813, 0.048704568296670914, 0.5241350531578064], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.27, 0.02, 0.71], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3c2fd5f89d9cbfc801061e4c3eb6b4e085e79befeec2a03f4bf3a80d4fdab9ce:action", "state_id": "bb9ab269b46c50a179e156db5eff4e4cae30a2dda662cd3d1c0ca7fa8f02c036", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.01, 0.98], "teacher_probs": [0.01, 0.01, 0.98], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4111328125, -1.087890625, 3.4921875], "student_probs": [0.019579043611884117, 0.009951288811862469, 0.9704697132110596], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.01, 0.98], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9090f43e0e78f8ded490b3715079403aa5971dde9712559b048ff3fb5b7e3dcd:action", "state_id": "997e7a5c10b1e4bb5cc8e04a218f77c6c8771a3db13d496e1679f07be28c2a8d", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.6900000000000001, 0.29], "teacher_probs": [0.02, 0.6900000000000001, 0.29], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.19140625, -0.11328125, -0.7158203125], "student_probs": [0.07483308762311935, 0.5978770852088928, 0.3272898495197296], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.6900000000000001, 0.29], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "63e84db2c32e829707db4feab3eede0460b8c7ecdad688451be3fca37e1528ea:action", "state_id": "4af36e9d2f7ab2b6dad8524be2bc2fa8cd58ea5bb0d746d9405d7443944e5722", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.15, 0.73, 0.12], "teacher_probs": [0.15, 0.73, 0.12], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.232421875, 0.91796875, -1.52734375], "student_probs": [0.22556328773498535, 0.7126506567001343, 0.06178612634539604], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.15, 0.73, 0.12], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4eb27e2de56ddf53ec6c1f1f620436eaf69ff9a859132161135670080a94d3bd:action", "state_id": "5f472f55f22d208bb04577836e75631328e16bf6df574429a2944396683b57ce", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.92, 0.03], "teacher_probs": [0.05, 0.92, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.140625, 1.11328125, -1.5390625], "student_probs": [0.2104826122522354, 0.7375319004058838, 0.05198553577065468], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.92, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "808793cc2fd34155818eb91ab2c9854127f301dd116a81c0f09f930712be7e5f:action", "state_id": "a8382528d11816058f23ab4090c69fbd6ada689f40804f76b722000fb3bf5d7d", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.99, 0.0], "teacher_probs": [0.01, 0.99, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.63671875, 4.310546875, -0.8251953125], "student_probs": [0.024609779939055443, 0.9696857333183289, 0.005704354960471392], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.99, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "055ec13a8740aa40eb4d5d6a14958035ebb320a80de6dc19763036c093f2acb9:action", "state_id": "37063b684f3ffddf781023f996683efc8d644128f7e509e660bfcef7e956c64d", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.5599999999999999, 0.41, 0.03], "teacher_probs": [0.5599999999999999, 0.41, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.49609375, 0.89453125, -1.78515625], "student_probs": [0.6307015419006348, 0.34559592604637146, 0.023702552542090416], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.5599999999999999, 0.41, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9b9f1d6c34b7eb1613b09f979dbbeee8d4a980d0472c7d2f94ecbdd0a7c94857:action", "state_id": "68fb3145ef4b36c2ac3f2b8e5976d9915ae4ca75bde6c21d92579b601ca2d981", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.45, 0.49, 0.06], "teacher_probs": [0.45, 0.49, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.5234375, 1.0078125, -1.22265625], "student_probs": [0.35744741559028625, 0.5801944136619568, 0.062358155846595764], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.45, 0.49, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "92490fd44cc48379a09a101bbcd7f6e1c454594ec7c72a0c2c8401d59106cbc4:action", "state_id": "4b86113a2bf9841714cb04765f8bbe46df5a70b2bc05382a8d573c34aa481a61", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.61, 0.33], "teacher_probs": [0.06, 0.61, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.78125, 1.11328125, 0.36328125], "student_probs": [0.03621474280953407, 0.6545824408531189, 0.3092028498649597], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.61, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5d4e539968d10edd6c5f4df3d5617f64ccd821b6157f098c8d7df9eac99c444e:action", "state_id": "75dfd01d4086119ba3c58ffafb5a815ee94b94d2482f6a7c268034dee91b26d9", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.43, 0.45, 0.12], "teacher_probs": [0.43, 0.45, 0.12], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.02734375, 0.4140625, -0.89599609375], "student_probs": [0.3362012803554535, 0.5227567553520203, 0.14104199409484863], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.43, 0.45, 0.12], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ca3886cdc16cd2779b99e4c36878c310066a152371bc70a5ba74c5e4f5cfcdf7:action", "state_id": "c3ecc5c225c9267e587e90bc1f9614b8609e902a804f003d49c945b25c4c36de", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.46, 0.49], "teacher_probs": [0.05, 0.46, 0.49], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.291015625, 0.31640625, 0.27734375], "student_probs": [0.09268958866596222, 0.4625145494937897, 0.4447959065437317], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.46, 0.49], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e23daa9a4091974462817e21a74c246456ed071f5b5f52a93330cd0e0181e391:action", "state_id": "597efb5335d6ce2d31ccf151d6b73805e9ec7bb8494ef8eddd7de767c7c0ae87", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.22, 0.05, 0.73], "teacher_probs": [0.22, 0.05, 0.73], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.544921875, -0.4921875, 2.58203125], "student_probs": [0.25307127833366394, 0.03300178796052933, 0.7139268517494202], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.22, 0.05, 0.73], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d866c66eafc2177d1a2fa543bd0322846679083a47a6069303a1dd1679f162bf:action", "state_id": "8cb3004d034d50e50289e0d9ce502b544b6becade68bc3f657f400e16b9dfd33", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.37, 0.04, 0.59], "teacher_probs": [0.37, 0.04, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.3662109375, 0.40625, 2.75927734375], "student_probs": [0.3813328742980957, 0.05371604487299919, 0.5649510622024536], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.37, 0.04, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "604841425991d2880b04d98d215dc4fd5f4134da85d8af2e23c27d1d857a2cf6:action", "state_id": "7b667d8311be3021d2931d393e48bd4c72a2be374e6c781db959db11f68f0548", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.89, 0.02, 0.09], "teacher_probs": [0.89, 0.02, 0.09], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [4.12921142578125, 0.05859375, 1.896484375], "student_probs": [0.8894404172897339, 0.01517994049936533, 0.09537967294454575], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.89, 0.02, 0.09], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "51fa4acd1210e813f0fb791c556e6a5b71509393af3f36b84f9be98917212944:action", "state_id": "6a68b6d7e6e14c1bbc282122ce330848ebff19725fd111b9c34f157a94717eb6", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.21, 0.43, 0.36], "teacher_probs": [0.21, 0.43, 0.36], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.119140625, -0.2265625, -0.109375], "student_probs": [0.16164560616016388, 0.3946441411972046, 0.4437103271484375], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.21, 0.43, 0.36], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fcadf1bc84bf6a94d4dec1c64caa39d8afc2c6b6377e0a196969cce150d6fa18:action", "state_id": "4a4a576e6466768227e1b5ed0b73586e89ee1dc09ba115d50ff53e06ce9c83f3", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.51, 0.1, 0.39], "teacher_probs": [0.51, 0.1, 0.39], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.40234375, -1.52734375, -0.39453125], "student_probs": [0.2163519412279129, 0.19092991948127747, 0.5927181243896484], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.51, 0.1, 0.39], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4f1b572c60d1eee195f9dfcdeee63bd2808f01cac64574461bbb3bc3751c68ac:action", "state_id": "cec63e6dc09790aa1563b27ecb352d68707de649f62a3454b7aafd78ff9aec54", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.48, 0.04, 0.48], "teacher_probs": [0.48, 0.04, 0.48], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.7587890625, -2.72265625, -0.7978515625], "student_probs": [0.47573620080947876, 0.06675280630588531, 0.4575110375881195], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.48, 0.04, 0.48], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4654681111e15fd89c226752635fc59d1dcda56572319d8a8fdba7b186c90081:action", "state_id": "1e3d6d41d409c78bea9abe069a20f8d2a9aa080777d2a18df960b0e23cf988d2", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.48, 0.04, 0.48], "teacher_probs": [0.48, 0.04, 0.48], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.880859375, -3.07421875, -1.33203125], "student_probs": [0.5719440579414368, 0.06379544734954834, 0.3642604947090149], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.48, 0.04, 0.48], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "33c9854abb21e7cc5a6fc036df421f214a95bcb4a2b2af4a95d692ba3f0ba424:action", "state_id": "fafde744bb0b503820eaef4126aaf5197eb115de76ac6c5d78d511f8245d8076", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.3, 0.02, 0.68], "teacher_probs": [0.3, 0.02, 0.68], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.1953125, -2.51171875, -1.02734375], "student_probs": [0.4079972505569458, 0.10938286036252975, 0.4826198220252991], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.3, 0.02, 0.68], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cd952f0ca3d60e70266965de8be84a2ed06172ab5b94ea518a3f2def2d2f2690:action", "state_id": "c6f1aa8e7f9a3ff6a3a6f04dae2ee76032898fecd640a5576ba90da006cf04e3", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.11, 0.48, 0.41], "teacher_probs": [0.11, 0.48, 0.41], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.05078125, 0.109375, -0.8271484375], "student_probs": [0.076499342918396, 0.6634399890899658, 0.2600606679916382], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.11, 0.48, 0.41], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f09ba1f0641025e9170e42d3461938a32c3d24b522fa4f7f5d6dc50d53da5a70:action", "state_id": "a8a9b89b50b58a37527fe0702c5e48285f640989899de34c4bb00365d99eb9d8", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.51, 0.43], "teacher_probs": [0.06, 0.51, 0.43], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.84375, 0.4453125, -0.5960693359375], "student_probs": [0.06969641149044037, 0.6876027584075928, 0.24270081520080566], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.51, 0.43], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "48d175ccd7d2c398b178995ecbd1d352ee321d7e8d18d96aedad8e4a5d8737ee:action", "state_id": "aa4ddc5cf6c5669eb8eabb2fe61ee4d122e3816c9140b082c06471c1df04cb49", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.56, 0.39], "teacher_probs": [0.05, 0.56, 0.39], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.33984375, 0.95703125, 0.390625], "student_probs": [0.06029050797224045, 0.5994722247123718, 0.3402373194694519], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.56, 0.39], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "20a9d84d6154c55a2eab2c5bd7c35f07405bfb131903fd8d4c39f387cf4b4a02:action", "state_id": "03a35895c898fbe70dfcc55c0a4019e6bce65e6d59d02379740ac25c3f4492c7", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.08, 0.59, 0.33], "teacher_probs": [0.08, 0.59, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.201171875, 0.91015625, 0.4375], "student_probs": [0.06940814852714539, 0.5732560157775879, 0.3573358654975891], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.08, 0.59, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2e7c7687c7efb40ae91b22a3716efd7914650343efbaafd4bc0032bc353a8faf:action", "state_id": "2e2d635b9e330813f14d4c7fceb0dde470475e9f34b3b986aba7e0aace678e1a", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.6, 0.33], "teacher_probs": [0.07, 0.6, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.55078125, 1.0, 0.4453125], "student_probs": [0.04722023382782936, 0.6052270531654358, 0.34755265712738037], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.6, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "880c46a181f83cf079066fab8c32352de44f5299d2fe1686c1eafb50b294cc94:action", "state_id": "7a83b918a2ecd71fef2ed3bd887ec06b923046b483a959826e8fa95678280eb9", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.44, 0.53], "teacher_probs": [0.03, 0.44, 0.53], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.62890625, 0.65625, 0.71875], "student_probs": [0.04697427898645401, 0.46162664890289307, 0.49139901995658875], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.03, 0.44, 0.53], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5db251450def53e74ad38b67451edca7e5260272a33744fa1b219c2502feaf61:action", "state_id": "55d42c23e90010feac917b1bcc43fb0c6ac191aeb219ba3d8f71640e2f196d5e", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.07, 0.86], "teacher_probs": [0.07, 0.07, 0.86], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.259765625, -1.119140625, 2.390625], "student_probs": [0.06417252123355865, 0.027172353118658066, 0.9086551666259766], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.07, 0.86], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c4b27e4f1412f65dad0724027c46f9eef96612dcb21c06cc3e59151c1f909a5d:action", "state_id": "f8934c230eba78c41a878f9b273fbb3dd74c0e409837e42082436ac511ef8162", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.05, 0.89], "teacher_probs": [0.06, 0.05, 0.89], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.0546875, -1.076171875, 2.6171875], "student_probs": [0.06318265199661255, 0.02274955064058304, 0.914067804813385], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.05, 0.89], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c1e4f8c1eea4cda70165d39d2985c03db09ac9751dc600727e42a6866cfd1b53:action", "state_id": "e55927dfcf702d29364209f18f9631eba5f12ac415910eb4bfe0bc9d0e827cd5", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.03, 0.91], "teacher_probs": [0.06, 0.03, 0.91], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.26953125, -0.94921875, 2.7109375], "student_probs": [0.07821797579526901, 0.023121187463402748, 0.8986608386039734], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.03, 0.91], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "816c5eeb83c4aade75668cc5fdc9e4e9840356260c1425b99c152b7467d2c7f6:action", "state_id": "a00fe790f30c8ac233d1dac5b46846c370899a6b818cdebdaa1126db37e3aa7d", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.02, 0.89], "teacher_probs": [0.09, 0.02, 0.89], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.1796875, -1.02734375, 3.1015625], "student_probs": [0.0503140352666378, 0.015048116445541382, 0.9346377849578857], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.09, 0.02, 0.89], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "45e3663678af2bb47af8e833ce844858796927be3dc85b53ace03aa5ac106264:action", "state_id": "d23a6af2a73f4a641b61615d3757d1010a64110bbab7cd3c18587251d908d290", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.01, 0.97], "teacher_probs": [0.02, 0.01, 0.97], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.240234375, -1.05859375, 4.064453125], "student_probs": [0.013247273862361908, 0.00584409898146987, 0.9809086322784424], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.01, 0.97], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cf0bbe5c008610848cda5ba94e95f7d75e24c364601e6a81bbe2f3dc33fbc2ab:action", "state_id": "b251d2c7885cb76dcf791a537394444b5d19303ca8f13663c1c2b4977874c236", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.75, 0.09, 0.16], "teacher_probs": [0.75, 0.09, 0.16], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.859375, -0.654296875, -0.513671875], "student_probs": [0.6786865592002869, 0.14937911927700043, 0.1719343066215515], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.75, 0.09, 0.16], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "dad14a24113020cd181c4cc2a90104186eccb9c93edb9765262f4cdce6c34249:action", "state_id": "99985a9d8a67042e29b8b86d45266e59f589bc44a7894d76db8a208cbdbe2f4e", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.62, 0.15, 0.23], "teacher_probs": [0.62, 0.15, 0.23], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.2265625, -0.533203125, -0.69287109375], "student_probs": [0.5357561111450195, 0.25061386823654175, 0.21362997591495514], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.62, 0.15, 0.23], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d43460a2b7880b17dd18e9511810bb9c78d9bd866a73ac3cf6c2229647496189:action", "state_id": "fd63a11ebd12584ba7698432ac1dde8e6ee40e00d1e0c2c0ad1fbcb6143cf1d0", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.57, 0.24, 0.19], "teacher_probs": [0.57, 0.24, 0.19], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.0625, -0.345703125, -0.619140625], "student_probs": [0.46069568395614624, 0.3062906563282013, 0.23301364481449127], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.57, 0.24, 0.19], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "386a643f239db386ea7ffc452c10384fa3a58730805bd37c05ec9737a67df53d:action", "state_id": "87484dd23187c61c141b176e88e7f97ddefd1427c898aa2fe51707834220af3f", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.12, 0.39, 0.49], "teacher_probs": [0.12, 0.39, 0.49], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.55859375, -0.42578125, -0.5263671875], "student_probs": [0.14468252658843994, 0.44914886355400085, 0.4061686396598816], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.39, 0.49], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b307076a021e0b7368cc78a5cd7c2ee239c988241208b2a9d36a5ce3248769ae:action", "state_id": "ca2d696835bb9c010f8c3238ad5395c234023f624e9ebebca5b841deb88ff795", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.72, 0.26], "teacher_probs": [0.02, 0.72, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.43359375, 0.51953125, -0.0390625], "student_probs": [0.01206293050199747, 0.6284535527229309, 0.359483540058136], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.72, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "69b0524f218f5126838cdb3fc028f969c6f72defc7ce70e71b0293b9bab82a81:action", "state_id": "b56a3c3f0f99b9f03c2fc62d5ed0eced70fe3b6956f5580a2faf0f711fad4808", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.72, 0.26], "teacher_probs": [0.02, 0.72, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.45703125, 0.6328125, 0.046875], "student_probs": [0.010641057975590229, 0.6355963349342346, 0.3537626266479492], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.72, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ddf8baad95a2b8a91936fd17576f6186bf73d87fe0e4da172595bad1a289a818:action", "state_id": "48fa7698d07e0585c3314bd49b456589d54fc3ed565a527bfc831c046dbdd528", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.15, 0.84], "teacher_probs": [0.01, 0.15, 0.84], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.8359375, -0.849609375, 0.78125], "student_probs": [0.008195259608328342, 0.16237100958824158, 0.829433798789978], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.15, 0.84], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "af5acd160dcc4ed2b1c5dd813c724f8268a3b792a70708143b2e81a5ac55e2d4:action", "state_id": "f79b05571f53cd8f81e64cdc318254aabb2b8e71b9d34ff19020b0d3e0640705", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.01, 0.93], "teacher_probs": [0.06, 0.01, 0.93], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.05078125, -2.82421875, 1.8046875], "student_probs": [0.13409963250160217, 0.008374116383492947, 0.8575262427330017], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.01, 0.93], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b0232dafe0e96716345c95bd34ea9a132aa43cd562a72a9e1f6fe04141fa7ac5:action", "state_id": "83818c80bbe90ebc785e9f7515c90bbebf511515a7a24cd112f12896e5c7fff7", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.01, 0.95], "teacher_probs": [0.04, 0.01, 0.95], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.078125, -2.3515625, 2.5], "student_probs": [0.08093869686126709, 0.007127813994884491, 0.9119334816932678], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.01, 0.95], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "aca04784b194317a7b0ed68a3699840fe491f76959a42b198eef623f612ca58b:action", "state_id": "71220dddb00065135044f95b4044da7212a167372cb56717273e1c51fc647793", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.0, 0.95], "teacher_probs": [0.05, 0.0, 0.95], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.328125, -1.953125, 2.724609375], "student_probs": [0.08273592591285706, 0.008452006615698338, 0.9088120460510254], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.0, 0.95], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "26011916b8454e5bf6c271e9b7f357a572dcc670da4d056f4e1fcb88819bea51:action", "state_id": "e60bb5ac9f0858222208c452ed080a9a51089f6d63c778f320ef752711cf62ee", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.01, 0.9], "teacher_probs": [0.09, 0.01, 0.9], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.328125, -2.16015625, 2.578125], "student_probs": [0.09460031986236572, 0.007856802083551884, 0.8975428938865662], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.09, 0.01, 0.9], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2cf19f0b48290a4a4d37a11f16764a71d0e1ada8b61af46618d5bbcfe0bbf7dc:action", "state_id": "7657c2fc30302c1d6c7d3d2930cb0ab4e7c88f9904a1d2dda1e0b3eba14a8a78", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.01, 0.9], "teacher_probs": [0.09, 0.01, 0.9], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.3125, -2.41015625, 2.859375], "student_probs": [0.07229171693325043, 0.004749566316604614, 0.9229587912559509], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.09, 0.01, 0.9], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cd877cc9d136779e53a6c834f047648ab80d90bf334c131b5877cad7bb9dc7ea:action", "state_id": "5e788ca2184144b438d32e745d55c1ff5ea7286a8bd56ed98da91b5ca2ea8451", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.0, 0.96], "teacher_probs": [0.04, 0.0, 0.96], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.03125, -2.49609375, 3.7578125], "student_probs": [0.023465391248464584, 0.0018742019310593605, 0.9746604561805725], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.0, 0.96], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "030ecf04d9d327f5705477904b6510b60fa21881abb8336bdf464594d06ccd31:action", "state_id": "9e1923dc8dba27a37cb3e44dc8f578f488b06fb36f13f068cb57c7c127e742b1", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.97, 0.01, 0.02], "teacher_probs": [0.97, 0.01, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8125, -4.2421875, -2.59375], "student_probs": [0.8327611684799194, 0.02697901427745819, 0.14025986194610596], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.97, 0.01, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0a7c983615facd8eb6a3121e89670694182ebd3dedc7f6f93d080388ea8b73f9:action", "state_id": "ac01cd94d493203f603a6f3b13481e6e481a175634a51c142759e6f9f20928d8", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.39, 0.6, 0.01], "teacher_probs": [0.39, 0.6, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.83984375, -0.775390625, -4.09375], "student_probs": [0.4750145673751831, 0.5066389441490173, 0.018346508964896202], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.39, 0.6, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "052eefab01e4aa6dd99a1ad3edbe034ea536252f2776f7b73c66b96ad8a80798:action", "state_id": "ab2a1505b530300e90f407fc94b52c897e8968abb4d0682adc170a4519261e66", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.87, 0.13, 0.0], "teacher_probs": [0.87, 0.13, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4384765625, -0.8671875, -3.8671875], "student_probs": [0.5939029455184937, 0.38683760166168213, 0.019259508699178696], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.87, 0.13, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ba8a122d1e88e625798e721b5eefa6bf8844271bc7e86ea49eb0e47a8880dd1d:action", "state_id": "23df87b9a2299d92defb67bd198d73a27a2ea1171d460815d56a806903e636da", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.93, 0.05, 0.02], "teacher_probs": [0.93, 0.05, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.275390625, -0.1640625, -1.3046875], "student_probs": [0.8967950344085693, 0.07820817828178406, 0.024996833875775337], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.93, 0.05, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6fe27d247cfaacae953fe35cb77e4a5b367b95acc1035173db01a0a4cf2ecb97:action", "state_id": "a51f74f9813832aea0a7400b8bf02f6a94559c4e8bec502275bd5a7ff13c1685", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.96, 0.03, 0.01], "teacher_probs": [0.96, 0.03, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [3.083984375, 0.4765625, -1.265625], "student_probs": [0.9202712178230286, 0.06784640997648239, 0.011882408522069454], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.96, 0.03, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7055c3f905672d6084a06b91fde1f5e0cd22161beaef1f1931713c9187d8e76c:action", "state_id": "3ae8bcd27a4041cdf4f52bfa0c29bc74970864283ac13ea2df2973b927c6b008", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.99, 0.01, 0.0], "teacher_probs": [0.99, 0.01, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [4.5947265625, 1.04296875, -0.73193359375], "student_probs": [0.9675535559654236, 0.027743814513087273, 0.004702576901763678], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.99, 0.01, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a98bd48ab3d7a21a77b5e1aa547bbbfbb027190aca050f1866e94376ffe0d110:action", "state_id": "b5ec76f1cd61d7d3fa8ee739a0e8d6ff93986348a742dc6a961dfd1a47572641", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.7, 0.29, 0.01], "teacher_probs": [0.7, 0.29, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.30078125, 1.4375, -1.71875], "student_probs": [0.6945711970329285, 0.29295334219932556, 0.012475458905100822], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.7, 0.29, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "98dc09a9e716908c365f633219820fbc8b71e0ebf1d5fefd0f63dd21141f83ef:action", "state_id": "f388a5aa1cb5d7309218d9d708f8d3090d8db443d4ec24e7e3079a677647f532", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.35, 0.63, 0.02], "teacher_probs": [0.35, 0.63, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.04296875, 1.611328125, -1.63671875], "student_probs": [0.35286402702331543, 0.6229349970817566, 0.024201033636927605], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.35, 0.63, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3ba5043b5565cedf7020a1e415add6b77a932543f7e193f709d1143d4d3f5f4b:action", "state_id": "31378a8dd9a308e54470d3b38779508045a284087ee9b199f68403c79d61c2bc", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.98, 0.01], "teacher_probs": [0.01, 0.98, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.79296875, 3.05859375, -1.703125], "student_probs": [0.0076902881264686584, 0.9838964939117432, 0.00841320026665926], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.98, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9796b0bafced7211e5c4a9156019bfbc348c5bba8993d930533e1e440387dabe:action", "state_id": "7e13d93d98cde3349b619e71859e3a83a699902483b3ac023e1cf99c3e2a2e99", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.0, 1.0, 0.0], "teacher_probs": [0.0, 1.0, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.94921875, 4.462890625, -0.85546875], "student_probs": [0.004420825280249119, 0.9907237887382507, 0.004855326842516661], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.0, 1.0, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ed9a19da9d71bbdc6fd82d9c1f7245d9fa561c4263793dbf5744911cb9d72d9e:action", "state_id": "91c8c6b0b4f2bd6d8a3892d371637744911aa47df85e9284dd4bbe136f23bb1f", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.0, 1.0, 0.0], "teacher_probs": [0.0, 1.0, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.171875, 4.87109375, -1.22265625], "student_probs": [0.0023635525722056627, 0.9953899383544922, 0.002246525138616562], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.0, 1.0, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e1b56f1612b5ccc2813292138daa622d3d3ee5342460dc2b6dbc542f135a946a:action", "state_id": "cc02e360de53732c27019aec1d868b282177221b2abf6de48d991a2457caab3c", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.8, 0.09, 0.11], "teacher_probs": [0.8, 0.09, 0.11], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.294921875, -2.41796875, -3.265625], "student_probs": [0.854019284248352, 0.10219746828079224, 0.043783221393823624], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.8, 0.09, 0.11], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "752bc141c03adf7fc22fb86faf60a006df4ea652584f1ea2e25647c8c365b21a:action", "state_id": "4c873ae5119471f4c2236258bc25fcf7d9bac474df559a2a518b86e8ede297ed", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.41, 0.03, 0.56], "teacher_probs": [0.41, 0.03, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.271484375, -2.546875, 0.23828125], "student_probs": [0.36131638288497925, 0.037127699702978134, 0.6015559434890747], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.03, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "08f94f1eae914502f623298d2894ebce1bfd68a2f45bbe10394b71449946993d:action", "state_id": "a37d9eef87792359bbf19080cfb8907098ebf524ff010851b65be4767bf7b88e", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.23, 0.71, 0.06], "teacher_probs": [0.23, 0.71, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.6737060546875, 0.71875, -1.5859375], "student_probs": [0.1842859387397766, 0.7416998147964478, 0.07401420921087265], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.23, 0.71, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9d2f6b4d31e37e3d3862c3a632d824cd673b1271f52ca69909772abc517b169a:action", "state_id": "e5876264c472093d7165273617d1fa3741ebfc4b7c5cc2fde4777b84a9d4c747", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.11, 0.87, 0.02], "teacher_probs": [0.11, 0.87, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.33203125, 1.4453125, -1.138671875], "student_probs": [0.13586068153381348, 0.8034971952438354, 0.060642093420028687], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.11, 0.87, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9cdc2f9eaacc383f4a3feb5f58576d861ac5e75d2f41086cacacd8c5bc2b4090:action", "state_id": "324b3cdc0b2d6c0ff7de1730225627cfdc4cf2f105ce347da98d9ad4c3650b9a", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.91, 0.03], "teacher_probs": [0.06, 0.91, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.1796875, 1.4375, -0.830322265625], "student_probs": [0.15242478251457214, 0.7680529952049255, 0.07952221482992172], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.91, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "95eeaac76639193b6185c1a101ac8b7e6b79d15c26d8419bfdb62160798a4db5:action", "state_id": "0b43a3e7b568ef989590504e4483b7f7486cc60fd336c421458aef51fac81aa6", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.95, 0.02], "teacher_probs": [0.03, 0.95, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.21484375, 1.646484375, -0.6533203125], "student_probs": [0.12380386143922806, 0.7963403463363647, 0.07985576242208481], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.03, 0.95, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c9a73d89aed5f5dac64dc5c6d78bb341542292c9efb31b51756ca8cfa1dee1de:action", "state_id": "165f1ac65c7a618f82a2277791ff77d833f7fc5bac2ea34a4fb774accb3889b2", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.98, 0.01], "teacher_probs": [0.01, 0.98, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4140625, 1.81640625, -0.829833984375], "student_probs": [0.0912071093916893, 0.84861159324646, 0.060181282460689545], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.98, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fced09de237116cdb957942378b2e693f061bec97844b8294d5f8260a0f9386f:action", "state_id": "9bb6fe0ce994aee7c6b1e82e8f93896aa15cf9214423a250ad57e5c42cb1f85a", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.0, 1.0, 0.0], "teacher_probs": [0.0, 1.0, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.08203125, 4.44482421875, -0.578125], "student_probs": [0.010629675351083279, 0.9828977584838867, 0.00647245766595006], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.0, 1.0, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "13170e04480819f3ed306a08c3b80e78b6d5863fa2a7a1e69ed1cfe4dc3f3300:action", "state_id": "d2261c8d8a419c660c818bfa56c1c87bd881018f08781a94654e2a9baacac186", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.64, 0.34, 0.02], "teacher_probs": [0.64, 0.34, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.890625, 0.85546875, -1.82421875], "student_probs": [0.492205947637558, 0.4752024710178375, 0.032591562718153], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.64, 0.34, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6e89e06ac16b8a08b56ac2a6844f9dd5dedd2d725b0602dc56211412f79318a4:action", "state_id": "c83978dec1c047949ee93f2ed3f60249c1ccf4d62da8997ec7226b8eead0f57a", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.47, 0.5, 0.03], "teacher_probs": [0.47, 0.5, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.171875, 0.80859375, -1.375], "student_probs": [0.32225003838539124, 0.6091390252113342, 0.06861098855733871], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.47, 0.5, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4b9e7f5bf4a6e738ce1dd046166568ca747b1a8ebccfb86053a5ba6679b7e9e5:action", "state_id": "794b83fe03f933d91c411411f803a59dcae2ae1e2bf8ae4ab70938c62d22d02f", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.79, 0.14], "teacher_probs": [0.07, 0.79, 0.14], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.375, 1.0, -0.4482421875], "student_probs": [0.026960359886288643, 0.7878972291946411, 0.18514244258403778], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.79, 0.14], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "51fde17a83bfcf77e86415a72b59a46b379d86102a4dbc24cac1bba8e962d442:action", "state_id": "38c38b034fc97a45de533178920f15509bc89f5ddec8dc3e0b53632492ce488c", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.49, 0.42, 0.09], "teacher_probs": [0.49, 0.42, 0.09], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.30859375, 0.42578125, -1.263671875], "student_probs": [0.2882707417011261, 0.6008078455924988, 0.11092142760753632], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.42, 0.09], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7a0712e189d0d498ab5a5c6204a0ca960926015a4803d02bdf640e15b840e496:action", "state_id": "113a87fe6bcd02061180019d8b755543b5afd30f0fc91624f3cc1589ce3b5d7b", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.09, 0.66, 0.25], "teacher_probs": [0.09, 0.66, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.0390625, 0.42578125, -1.19921875], "student_probs": [0.06632333993911743, 0.780071496963501, 0.15360519289970398], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.09, 0.66, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e91a88cb4f4c694ec595827b6f7ef53f2d42615b1b05e84c91c35e4f03b6ed45:action", "state_id": "77dfd4f1ba0febf772184e7f659dd6659e44ec0acab5f41535833ed62f754163", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.66, 0.32], "teacher_probs": [0.02, 0.66, 0.32], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-4.375, 0.09765625, -0.6169662475585938], "student_probs": [0.007607274688780308, 0.6663141250610352, 0.3260786831378937], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.66, 0.32], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bfdd8e9736bcf48712f22fc2defb2d061d6b0f8b123eca1f41ac370a86801809:action", "state_id": "0b79942042364aae51f382b96ae9fc70d810d0d064853732dc3259f28e627c1b", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.44, 0.55], "teacher_probs": [0.01, 0.44, 0.55], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-4.546875, -0.00390625, 0.16796875], "student_probs": [0.004841191694140434, 0.4549236595630646, 0.5402352213859558], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.44, 0.55], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2453095742505adb3e84c9eae00404196ee17ad702da43f15731be1874ddd1cc:action", "state_id": "b1ba72e3a24aef899b37a7799a45a9986e7c8bf1af058d695fc61b0fd9b29da8", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.24, 0.1, 0.66], "teacher_probs": [0.24, 0.1, 0.66], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.41015625, -1.13671875, 1.341796875], "student_probs": [0.2665541470050812, 0.05675265192985535, 0.6766932010650635], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.24, 0.1, 0.66], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "186fb29df38809d6fc44f631e16cde8b9cc0bf0433976e0e186d46deba884a69:action", "state_id": "8d2bcb5f94d8575f3e8c98630251b156c0c42eaa01c71dfb1a097121a8bae04c", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.25, 0.09, 0.66], "teacher_probs": [0.25, 0.09, 0.66], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.056640625, -0.54296875, 2.0234375], "student_probs": [0.260995477437973, 0.052714668214321136, 0.6862897872924805], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.25, 0.09, 0.66], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1defa53771a8db070c258b4b3fc7ee31319a3de380f4191d63a299de7f89b8b0:action", "state_id": "c31bacf50b348647277456f34565aae38e793e46c51a15d8edaba6cb12473a82", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.29, 0.05, 0.66], "teacher_probs": [0.29, 0.05, 0.66], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.51171875, -0.267578125, 2.4560546875], "student_probs": [0.26738953590393066, 0.04512379318475723, 0.6874867081642151], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.29, 0.05, 0.66], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "61e6d5322fcf54bba068c4e00a3b9be8a0de652c45ce6b20635833f6e8d556af:action", "state_id": "f51ad215cc27a303500856a9905e4b970dc945e99e19db856f877e027f6da6cd", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.41, 0.06, 0.53], "teacher_probs": [0.41, 0.06, 0.53], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.4375, -0.08203125, 2.193359375], "student_probs": [0.29866302013397217, 0.06535177677869797, 0.6359851956367493], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.06, 0.53], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "207a6fa363455303122e6d924427cf8730f77c972634dadfd2a332bbc16d401c:action", "state_id": "91ed3857c61fdbbb02294a64b3dd427058d36acdd9ecbef89737b9d18b3cdde4", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.35, 0.05, 0.6], "teacher_probs": [0.35, 0.05, 0.6], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.16015625, -0.287109375, 1.71875], "student_probs": [0.5781478881835938, 0.05002705752849579, 0.37182503938674927], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.35, 0.05, 0.6], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e0efaee445aea1a1b9f41ddda426fb811901b9a98ead112b2269626eeed9798d:action", "state_id": "de8fefa384e7565823c0862f708268101bf1328db32199e3473287ca9a9ee35d", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.92, 0.02, 0.06], "teacher_probs": [0.92, 0.02, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [3.798828125, -0.6939697265625, 0.65625], "student_probs": [0.9484419822692871, 0.010612396523356438, 0.040945522487163544], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.92, 0.02, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8cbd3869fbcfc8ea995d0f0e1402c54050cbb907d120fe09bee716e572806d7e:action", "state_id": "3f4f417a8c4c1d7114d5bbb0b1857ee30a7e07b7b3927c278f690f4f9775717c", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.68, 0.16, 0.16], "teacher_probs": [0.68, 0.16, 0.16], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.59765625, -2.125, -3.18359375], "student_probs": [0.5571259260177612, 0.3287992775440216, 0.11407473683357239], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.68, 0.16, 0.16], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fd0bf229ef54498d4b6029e6a3334b84e487569d954a1a585785e49e91773a8d:action", "state_id": "316dc4048105492151c0e7006df30f1e67dabe745772dc9cf24b5e07ad34ada6", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.58, 0.03, 0.39], "teacher_probs": [0.58, 0.03, 0.39], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.5283203125, -3.09375, -0.9375], "student_probs": [0.5743558406829834, 0.04416000470519066, 0.3814842104911804], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.58, 0.03, 0.39], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "475107659620f18151a99a49ee037d11758efdb946534c2d656e6873b8f53b85:action", "state_id": "39fa94c3160fcbd679f821887756c746cb45811534befcc381355c835174e2fa", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.5, 0.03, 0.47], "teacher_probs": [0.5, 0.03, 0.47], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4013671875, -2.94921875, -0.7646484375], "student_probs": [0.5638121366500854, 0.04411807656288147, 0.3920697867870331], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.5, 0.03, 0.47], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3137b10492357d43991b2b1c1bc398a8222c2f6b70ff0abf057bf93fc1077274:action", "state_id": "58d8dcba5deb6a675617fb5fe12f04c23577f74811ec6a0487977ffa5962949f", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.52, 0.02, 0.46], "teacher_probs": [0.52, 0.02, 0.46], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.6220703125, -2.6875, -0.15625], "student_probs": [0.3676356375217438, 0.046602893620729446, 0.5857614874839783], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.52, 0.02, 0.46], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b5393d11c6da98b00ea679643ed635435a71f7a5325364c2e67849e4e290ebc0:action", "state_id": "a69bd6ba236bd272ee28e086983be1c13f983954fafe5bda942c95eb0f58f197", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.41, 0.03, 0.56], "teacher_probs": [0.41, 0.03, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.53759765625, -2.7265625, -0.4580078125], "student_probs": [0.45560672879219055, 0.0510428324341774, 0.49335047602653503], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.03, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a3d321115b60f0814de3e47a07c495041d34d02ce3ced6e181e51179d0f118a2:action", "state_id": "aad560cc7c6499621ca1311003b7e8ca19ad9dabe923d5d2d8426d6b1a9d1317", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.58, 0.38], "teacher_probs": [0.04, 0.58, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.69921875, 0.765625, -0.4091796875], "student_probs": [0.06099579855799675, 0.7174108624458313, 0.22159337997436523], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.58, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b184b421663a85c3b5e0a3f4e99bc88eb69d2dd35e0d64f6e74d1fc808705999:action", "state_id": "c32404c6b71219a719a2d20e54419849a45bf16e4574dd2f4518841d309579ee", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.57, 0.38], "teacher_probs": [0.05, 0.57, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.421875, 1.01953125, 0.15234375], "student_probs": [0.057749539613723755, 0.6634951829910278, 0.27875521779060364], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.57, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "024027960eea6a063565b01fe9dda747e31d4bb09355ef49ca5bf5006a8c8d3a:action", "state_id": "365c179c33f51874ed92edfc49ef2f8261a77b63811e2b173c1716ccbbdb0ee8", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.38, 0.61], "teacher_probs": [0.01, 0.38, 0.61], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.84765625, 0.97265625, 1.5], "student_probs": [0.021636545658111572, 0.36310651898384094, 0.6152569651603699], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.38, 0.61], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0ae0b307c6e701c46c8d60a94454cbd3b3ba4ebcb4ed5e54cc2b16b776e51266:action", "state_id": "aea171a084b983c5a4a22c93cf0e14b4f97e19c70c2f59977fcfe067f6a4ff04", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.04, 0.92], "teacher_probs": [0.04, 0.04, 0.92], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.15234375, -0.66943359375, 2.955078125], "student_probs": [0.04173697903752327, 0.024885809049010277, 0.933377206325531], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.04, 0.92], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "dd513f7f7d9f62b79e7968dae97471b28dbecc549011633ad9ad71453435f3d9:action", "state_id": "4f5b5ce0109c26f8e2d304fb40f3aac64a68340e19eabbf83c780b1b48b3a4e9", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.03, 0.92], "teacher_probs": [0.05, 0.03, 0.92], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.17578125, -0.5849609375, 2.826171875], "student_probs": [0.06399226933717728, 0.02990483120083809, 0.9061028957366943], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.03, 0.92], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bbcf5e062fb9a0d7aec7a0685631c30d8c90bcfbacf9fb8543dbd4937fb12e0f:action", "state_id": "ff366f7a58cca8d10ac15ce18e46fe3d09411ac07fffaad4e0c9fff9ea12b7ef", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.0, 0.99], "teacher_probs": [0.01, 0.0, 0.99], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.5546875, -0.4716796875, 4.1142578125], "student_probs": [0.02739246003329754, 0.009814889170229435, 0.9627926349639893], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.0, 0.99], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ae9c59b0f1d6562dceade71719564d6154280b1137fe7e0ab6af9d269ce4e351:action", "state_id": "663d365da04c525b2e046ffa7ade694592c3c37986f0b6dd48fb84b99792c2d2", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.78, 0.06, 0.16], "teacher_probs": [0.78, 0.06, 0.16], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.08984375, -0.8857421875, 0.25390625], "student_probs": [0.8261176347732544, 0.042146481573581696, 0.13173598051071167], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.78, 0.06, 0.16], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "262ead12904342099e506cf10db5f7ebe4a1abde64e3cacea960785b5c1c6131:action", "state_id": "c3a74d710945b6330fa6bb7b6d3b293792e0f90520f764f162192f6f928075ac", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.1, 0.87, 0.03], "teacher_probs": [0.1, 0.87, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.0390625, 0.00390625, -1.83984375], "student_probs": [0.23328474164009094, 0.6619755029678345, 0.1047397330403328], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.1, 0.87, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8912c9f2b7e1b87e67fd802dae3a220fb63862ddd45c50bfbb19f8f51d53ef82:action", "state_id": "299afa369bc48213d303163c8dd47ac1361afc6e2eb37072221c4846271a86b8", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.12, 0.86, 0.02], "teacher_probs": [0.12, 0.86, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.23828125, -0.203125, -1.5546875], "student_probs": [0.22005543112754822, 0.6195762157440186, 0.16036833822727203], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.86, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "387ac6e9bc63ab2bdb01df5ad4964b9a88918a5c21632d6edd60b08b5c4f4273:action", "state_id": "80d3db80e4c9d35a74a600e5e6ba519f6c77d7672303e5d60e23482bce4a8d06", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.3, 0.63, 0.07], "teacher_probs": [0.3, 0.63, 0.07], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.3359375, -0.908203125, -2.46484375], "student_probs": [0.34999722242355347, 0.536818265914917, 0.11318447440862656], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.3, 0.63, 0.07], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6fd7853e414d025d1ada4b27f8b694247274ae30ca5025232818b902ccede411:action", "state_id": "6d81a23831438030ca55ebadf063a52f4446d64d540c9625679a5befe2173e61", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.59, 0.3, 0.11], "teacher_probs": [0.59, 0.3, 0.11], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.61328125, -2.4296875, -2.8515625], "student_probs": [0.5774007439613342, 0.2552211284637451, 0.16737809777259827], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.59, 0.3, 0.11], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f65cd7e5a98cfe662e471c66418383c5c5f66f7cf3d67d1ff5b7de47c933c3f9:action", "state_id": "51fe585f5a8fedb509f991f5ee24aa59c8379044a7739768d00a3a7ab4547fc9", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.6, 0.02, 0.38], "teacher_probs": [0.6, 0.02, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.65869140625, -3.37890625, -0.59222412109375], "student_probs": [0.468474805355072, 0.030854033306241035, 0.5006712079048157], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.6, 0.02, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "240ff3a3b80d47f4cd9496394304dc055dcb7238b6580672673eafccdad08b17:action", "state_id": "aba4cb2f427aedd44202ca2a5f6dadcc6ced88760c17aee001c9c63ec8f9b151", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.44, 0.03, 0.53], "teacher_probs": [0.44, 0.03, 0.53], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8525390625, -2.87890625, -0.3955078125], "student_probs": [0.3688414990901947, 0.04861828684806824, 0.5825402140617371], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.44, 0.03, 0.53], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "41a2d4cd37a24dbd5da8599293ce5b3ed08479e702fb52745ab8f10ae605b9a3:action", "state_id": "fe1c838c6986c8e4490a6b1b3b32a65f37db98547e2721a2fd359f22cf759c18", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.73, 0.21], "teacher_probs": [0.06, 0.73, 0.21], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.75390625, 0.2109375, -0.357421875], "student_probs": [0.0821371078491211, 0.5859495401382446, 0.33191347122192383], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.73, 0.21], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9bcb489ef3c301c60c7b5fb42312b6a4a0028ca0961e405f2fc55f67731da27e:action", "state_id": "86f675dbf942243b918bcc48b2ccd06b16f20ef808e34b7dc89ebd976ea68d61", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.62, 0.35], "teacher_probs": [0.03, 0.62, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.138671875, 0.4296875, 0.3046875], "student_probs": [0.09966444969177246, 0.4782666563987732, 0.42206883430480957], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.03, 0.62, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0b490659f420558524f3e411f066293aba8e0bdd8535e881c5d0d91a20807ea0:action", "state_id": "01c56ffa73f94acd9fe8ffa23f261b1009af80c52401ccac8fae2bd1ce15f901", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.64, 0.35], "teacher_probs": [0.01, 0.64, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.234375, 0.96875, 0.69921875], "student_probs": [0.05893593654036522, 0.533562421798706, 0.40750160813331604], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.64, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cb035f034cdcec9ea2ccf4bc04626f6f41c59aada2ad9675179685f94de77382:action", "state_id": "fa4bf862e18cb3a9f331f31ecfd310f26f36508d30706c967968ca1b9ed0df9b", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.63, 0.36], "teacher_probs": [0.01, 0.63, 0.36], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.98046875, 1.1875, 1.01953125], "student_probs": [0.058378588408231735, 0.5102587342262268, 0.43136265873908997], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.63, 0.36], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7318770f1f1a5bf118c2b61077e5596235eed47b27019bcda39b7541fe19ed49:action", "state_id": "e3ac539f9348ee7c89d8a6a728ac525886724ebeafcc9bef99c12cce9b17d066", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.57, 0.42], "teacher_probs": [0.01, 0.57, 0.42], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.23828125, 1.28515625, 1.271484375], "student_probs": [0.03879963234066963, 0.4838854968547821, 0.47731488943099976], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.57, 0.42], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "895355c4e340396ae9f1f14b74578ee2672a8b639dcbeb801e1428cd26c56bed:action", "state_id": "19aef764997fe53408cec05b136ed3528b0643c08c756f084beed7d5efb3671c", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.29, 0.7], "teacher_probs": [0.01, 0.29, 0.7], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.51171875, 0.71875, 1.677734375], "student_probs": [0.028918974101543427, 0.2690686583518982, 0.7020123600959778], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.29, 0.7], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9da20b0ce9bf1c0c18f26a6690fc1e90a7f1d1c0f44c5238c68c4047586ba3dc:action", "state_id": "9558108649039c21596ac269e03a40576556b886bf20ab912f23615bea6b72b6", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.0, 0.99], "teacher_probs": [0.01, 0.0, 0.99], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.109375, -0.421875, 4.725341796875], "student_probs": [0.009739622473716736, 0.005725629162043333, 0.984534740447998], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.0, 0.99], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0fb27b063842a515d0911dcb24c7c9179d1472db40e82f50a811c7547f7d3621:action", "state_id": "ed5df0dbc23f20de2eaeb1a81b2071812553ab086f5ebd1d0edc957c09af22b7", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.36, 0.08, 0.56], "teacher_probs": [0.36, 0.08, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.55859375, -0.7589569091796875, 1.060546875], "student_probs": [0.3424968421459198, 0.09171737730503082, 0.565785825252533], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.36, 0.08, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b86ca048070b19568729b7abe92b68199e9691501d8e66b4e16c6000817a45c2:action", "state_id": "3cfccf5a6fd2a7a3f99d6aa9a9b651474a5374428db651cd9fa6b61e0343e364", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.53, 0.09, 0.38], "teacher_probs": [0.53, 0.09, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.73046875, -0.9716796875, 0.35546875], "student_probs": [0.5348792672157288, 0.09750392287969589, 0.36761680245399475], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.53, 0.09, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4469577474c8fd7841924f413f60cca2957db01a57e50b58a783cc8de7a8713d:action", "state_id": "ae07e05514403331f9b2dfb38a850c450fa9be9a876efaa0ac448eea4b9b6706", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.08, 0.86, 0.06], "teacher_probs": [0.08, 0.86, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.23046875, 1.08984375, -1.390625], "student_probs": [0.08311954140663147, 0.8460617065429688, 0.07081872969865799], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.08, 0.86, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c8f192cfba32b3feae3b0ca85c1b6a9fa34b89554716b1b499d74e5df52e230e:action", "state_id": "b47258a91b184d1b0e2a8b1cf49676b3802afcb19f27bef581a8ea212b89c77a", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.95, 0.02], "teacher_probs": [0.03, 0.95, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.248046875, 1.21484375, -1.27734375], "student_probs": [0.072940394282341, 0.8562250733375549, 0.07083447277545929], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.03, 0.95, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6a30b39d732b3d37b27e658ff3968e0fb3402e2f39ca9becd547cbf875e51514:action", "state_id": "9946e0b94f2733595479124ffc731dd36e456c7f03dbc605dacd3ca5b6b39d7f", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.98, 0.01], "teacher_probs": [0.01, 0.98, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.177734375, 4.076171875, -0.675537109375], "student_probs": [0.013891268521547318, 0.9776646494865417, 0.008444014005362988], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.98, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1346c5a81ffe10f8d6719807fd11d5868004a51b90df6616a19054cf185f8ae6:action", "state_id": "b08104dc8337d3bd0549b311bb017f764c37c7b61474db5162436761276d7be7", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.16, 0.16, 0.68], "teacher_probs": [0.16, 0.16, 0.68], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.30859375, -2.9296875, -2.21484375], "student_probs": [0.18361645936965942, 0.26820600032806396, 0.5481774806976318], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.16, 0.16, 0.68], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a27fc8354c730d9572c15f9d9b9bb5cc4048ab680d9caf88257e7bf03e2a3771:action", "state_id": "652182a0fe47b799a15e84d2f1046d62537ac696b7554cea4992a6f9444b5891", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.31, 0.69], "teacher_probs": [0.31, 0.69], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.93359375, -2.5703125], "student_probs": [0.4101655185222626, 0.5898345112800598], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.31, 0.69], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2377055066b1aa5c70107a7eac3578a5c3ff674f29cb66fe6f9480ab4599bf73:action", "state_id": "1996f099f738e43082ea1d9ad2a4ea7abc0b0ec3f4f9e8270df1717429de41b8", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.62, 0.25, 0.13], "teacher_probs": [0.62, 0.25, 0.13], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.4453125, -2.55859375, -2.92578125], "student_probs": [0.19575630128383636, 0.4751304090023041, 0.3291132152080536], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.62, 0.25, 0.13], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f258a8bae34c20bb7da8bf571dcd6d484dc65324a36738bf5a3d786b967b9e16:action", "state_id": "a086dca806ca4d9560971b1df2d36823825b3fa583bde321ec1a8c3b06d1cc5b", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.51, 0.31, 0.18], "teacher_probs": [0.51, 0.31, 0.18], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.97265625, -2.390625, -2.828125], "student_probs": [0.2534746527671814, 0.45363596081733704, 0.29288938641548157], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.51, 0.31, 0.18], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f0d5e583a7e01de342294e426d66b6a211ae60e6240b9e7f8c6eb84fd7995902:action", "state_id": "270f70521296f6b9ec921eae3e679d98dbee25a1715bfeaad2f3a12a78178648", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.44, 0.4, 0.16], "teacher_probs": [0.44, 0.4, 0.16], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.015625, -2.26953125, -2.8203125], "student_probs": [0.23124395310878754, 0.48763492703437805, 0.281121164560318], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.44, 0.4, 0.16], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c2fc714986583e02983c654186331652d13d3b33e6bd8d87dcfd32e03d771d05:action", "state_id": "30e43a1cebb7862d71b29c144c5f4a8d16ac3dc4e51295a195b7723707270036", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.42, 0.43, 0.15], "teacher_probs": [0.42, 0.43, 0.15], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.703125, -2.21484375, -2.6953125], "student_probs": [0.27492496371269226, 0.44799381494522095, 0.2770812213420868], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.42, 0.43, 0.15], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "dbc2214a903b7ff35ff487793d2f0408aced2c2b47cad4a64555bce8f6cfc568:action", "state_id": "141b3b5873fedfb0dedf42f80acba42e205a487db2667cf4208069ce3f44513d", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.42, 0.4, 0.18], "teacher_probs": [0.42, 0.4, 0.18], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.23828125, -2.203125, -2.64453125], "student_probs": [0.1777363121509552, 0.5004248023033142, 0.3218388855457306], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.42, 0.4, 0.18], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "65b9edd6267d6799c60ec40470298c5264c09f6db193316a2a01c4ffeccfc4eb:action", "state_id": "bec68620c18e6da9e03ba5426bd0d232cd3ed373ed26cc899f4982bcc512492f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.27, 0.73], "teacher_probs": [0.27, 0.73], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.890625, -2.48828125], "student_probs": [0.4007493853569031, 0.5992506742477417], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.27, 0.73], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8fb803f1a28560e6310bcc85a0348850c16f85b6bcf1e64c2ea15b8a1241cdc4:action", "state_id": "16217c51f2768bc7d8b21e94c0689143430f98a9c8c6bf119b9cb16a9308e389", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.22, 0.78], "teacher_probs": [0.22, 0.78], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.625, -2.09765625], "student_probs": [0.17838266491889954, 0.8216173648834229], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.22, 0.78], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b72f80aad47004a59d9514356e72482328948cd9589fc636a62c65794db7eab4:action", "state_id": "768aaaa9c10e0408dcdc0f4d49fb868777a9122bf350250bfd1b36a7d9339a42", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.27, 0.73], "teacher_probs": [0.27, 0.73], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.99609375, -1.8671875], "student_probs": [0.2443629950284958, 0.755636990070343], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.27, 0.73], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "28e01ac1a98150f2be38443978daa11f1d029aaa5c752f21177c7d3ac23f276e:action", "state_id": "2db7596a76bf15738910d2f9259086cea028c83f2929c9f7874ef899c778a482", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.51, 0.49], "teacher_probs": [0.51, 0.49], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.1328125, -1.85546875], "student_probs": [0.6731916666030884, 0.32680830359458923], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.51, 0.49], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "805e40cb1f1a689bc440450bdad8ed4061acdfd51aef1390d12e31ce1d6a3358:action", "state_id": "20a16f2f40ad6e41479c693d436c55e6b6bde386552d41fb834396e797bf3e49", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.31, 0.29, 0.4], "teacher_probs": [0.31, 0.29, 0.4], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.0419921875, -1.130859375, -0.90234375], "student_probs": [0.32628166675567627, 0.2985369563102722, 0.37518131732940674], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.31, 0.29, 0.4], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b90e02ee90e703ab42c8da11427ee7194c24257d4fd6afdd0744058acf78e660:action", "state_id": "78c94b401fda97473d8342955f43eb555186cde8f3d9a3f10edb8a3dad6eb370", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.66, 0.34], "teacher_probs": [0.66, 0.34], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.23046875, -1.11328125], "student_probs": [0.4707365930080414, 0.5292633771896362], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.66, 0.34], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f161a2fe05ba756a3d05bbe8944f2685b97a2211ef81fa5a3b0b304a2b421322:action", "state_id": "acb06af22cc6f204d8b28e79913fbb273a24631d0f103ac9f855cc2e82636c88", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.19, 0.81], "teacher_probs": [0.19, 0.81], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.51953125, -2.07421875], "student_probs": [0.3904758393764496, 0.6095241904258728], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.19, 0.81], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1efd34fe73fa5b835524f3336fabf56b44b461d6e5d114bad961cc049eb2cf5d:action", "state_id": "e771c418b3d9340f14c60d0d169f43504206bc9a3a83dcd56655d2de05688e03", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.3, 0.61], "teacher_probs": [0.09, 0.3, 0.61], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.5859375, -2.66015625, -1.72265625], "student_probs": [0.23259080946445465, 0.2159532606601715, 0.5514559745788574], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.09, 0.3, 0.61], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1231a4d38876dead54cc07bc682ba6df835d0233ebfe1e01ac9c03ee31eef4c4:action", "state_id": "472c7210935ddb3cf5389cf60a2fe9cbcf8eb21614866480294c56551f5070d4", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.44, 0.15, 0.41], "teacher_probs": [0.44, 0.15, 0.41], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.10546875, -2.7265625, -2.078125], "student_probs": [0.38985177874565125, 0.2094893604516983, 0.40065887570381165], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.44, 0.15, 0.41], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a601fae54e0cce7ec6770489bfd83be496ff2416422a2026762c0d90eb495d68:action", "state_id": "41b8f0a4edb250aed608bab431c1dbd5f99438ea3a432c6873d48b7c5e295edb", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.34, 0.66], "teacher_probs": [0.34, 0.66], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.86328125, -1.015625], "student_probs": [0.29992473125457764, 0.7000752091407776], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.34, 0.66], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ff4e17ab4ea1776f3df17e10aaf6245cefc9fb848a56b8a1994afa6716b58f42:action", "state_id": "a7fa3d0accf76fac9f9bb9762a62af0611425898525cccb9c00b923ffa47cfc7", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.35, 0.13, 0.52], "teacher_probs": [0.35, 0.13, 0.52], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.052734375, -1.91015625, -1.130859375], "student_probs": [0.42569437623023987, 0.18060272932052612, 0.3937029242515564], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.35, 0.13, 0.52], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5896880248f8edecf4c59e536d5694759752f67f23f08d601598bdd8d8227c52:action", "state_id": "5d674e767a0ef71dcb5120b1a14fc50355535773d4ff988f66064fc46180f1fd", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "west"], "gold_index": null, "teacher_raw_probs": [0.74, 0.26], "teacher_probs": [0.74, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.71044921875, -1.8203125], "student_probs": [0.7521036267280579, 0.24789638817310333], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.74, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8035f89086bdab9ab409f3bffa51ae07aa201218f6b5d5e3f0fe21aba58a44a4:action", "state_id": "7f5fbfdca47f406ba9186064210c9f132edb40a2e8ee5faaa87489a6620acfdb", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.24, 0.49, 0.27], "teacher_probs": [0.24, 0.49, 0.27], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.537109375, -0.7449951171875, -0.8671875], "student_probs": [0.39507460594177246, 0.3209190368652344, 0.2840062975883484], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.24, 0.49, 0.27], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2dde1eb82294b33a9568fe3a2b824c7d2abdc2ea82c32b79cc2fe49b7b1fabb8:action", "state_id": "b22c8af2d73618d5f5189faa64c09b84cd30bb694a5485eddabdbc12db92bbaa", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.45, 0.55], "teacher_probs": [0.45, 0.55], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.79296875, -1.22265625], "student_probs": [0.36116471886634827, 0.6388352513313293], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.45, 0.55], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c6c8381523ffbe7eb4f88dca71466d7948076b17d32a2fc5ba3d6d0c8ed78472:action", "state_id": "906705ca99cb94d29e46df3a8c14dbb44b762a84dca447d4815e341c388d42ff", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.42, 0.14, 0.44], "teacher_probs": [0.42, 0.14, 0.44], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.6015625, -0.13671875, 0.3125], "student_probs": [0.4490547180175781, 0.21461881697177887, 0.3363264501094818], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.42, 0.14, 0.44], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b1f25240706e0fe9d8982b72522ccd23e58b42dfed33e8bc70627af7beec78ef:action", "state_id": "f4c9e1de98bd789f1fc882b905fcb00a4d84e604f210990c668d36c71d93d0b2", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.32, 0.03, 0.56], "teacher_probs": [0.09, 0.32, 0.03, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.13671875, -1.53515625, -2.22265625, -1.8671875], "student_probs": [0.1979425698518753, 0.36123886704444885, 0.1816423088312149, 0.25917622447013855], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.09, 0.32, 0.03, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "43bc5f8d55fb6a938dd9ae7751bb742bff070263fdfb1d6d085a796b50498158:action", "state_id": "bd6e5b528f4d6e5f98d14755bfa49bef429353a2ced01ddfb8a56de92ebed5cf", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.47000000000000003, 0.2, 0.33], "teacher_probs": [0.47000000000000003, 0.2, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.96484375, -2.07421875, -2.875], "student_probs": [0.4349990487098694, 0.3899306356906891, 0.17507030069828033], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.47000000000000003, 0.2, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "49a0d0f8032d21674bb9d39635fc15700994f2cda15b0efb3e8029c39bf50a38:action", "state_id": "227d8caa9010b0ed7da1ca094e9cf6b075b0fe1f4f8846f81708b0183b0964dc", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.25, 0.42, 0.33], "teacher_probs": [0.25, 0.42, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.83984375, -1.5390625, -1.3828125], "student_probs": [0.4812333285808563, 0.2391601800918579, 0.27960655093193054], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.25, 0.42, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "238a14bc460199ac976345c727f67b37a4009504bb6eb800e97a48074996acaf:action", "state_id": "ea80b6b3849f99509dcc684bcf35b553a9d6cae1c1a731ecd8781f8cd1ee2e4f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.24, 0.76], "teacher_probs": [0.24, 0.76], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.55859375, -2.11328125], "student_probs": [0.6352224349975586, 0.3647775650024414], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.24, 0.76], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d5c9b8c87414f00d05a07a8038a5297782acdcb058fdbfd302994d8ae77de82d:action", "state_id": "2c04c39ef6881d71964fb369e537805e064aeca16aa5f820292accca01810122", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.35, 0.65], "teacher_probs": [0.35, 0.65], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.046875, -1.84375], "student_probs": [0.4493926465511322, 0.5506073832511902], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.35, 0.65], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "85739eb050c2fffe309a67df8a429c6bb2ebfe4fd03796c3b38de9656c2a4688:action", "state_id": "2a7f49a832b06cf11218cfb3db78b6657f3ddb78465b8a7a7c291d3f9066631a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.36, 0.17, 0.47], "teacher_probs": [0.36, 0.17, 0.47], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.55859375, -2.03515625, -0.79296875], "student_probs": [0.26516392827033997, 0.1646440178155899, 0.5701920390129089], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.36, 0.17, 0.47], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cb763aaf45790c6f6ca32f2764db6c109d08eea1720177bd7d8eea83f3243573:action", "state_id": "e369754f2b9cbbbcd58d05f07e8892d9219078bde48db2f12ab79e0ab9add3ab", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.14, 0.34, 0.52], "teacher_probs": [0.14, 0.34, 0.52], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.578125, -2.234375, -2.24609375], "student_probs": [0.2628796398639679, 0.37071970105171204, 0.36640068888664246], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.14, 0.34, 0.52], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2d9758c9d6e7cc3bf5011ad7f37a1d8670e36dc75b1b4b3d469ce373120ae3ae:action", "state_id": "af0a3a5fb02f8606d57b4b89a2fce4480b5bcd9f46fa18dc5e6d67ad01e940d0", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.24, 0.76], "teacher_probs": [0.24, 0.76], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-3.04296875, -2.91015625], "student_probs": [0.4668456017971039, 0.5331544280052185], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.24, 0.76], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "31a89457ce3d9a4bac3fe78ca8d7902396212eb0b32860268da2f0a09f64ce33:action", "state_id": "825e84d113cdeb4d9933ed19185fc9716b97412df7b845024c47d122afc99a3f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.4, 0.51], "teacher_probs": [0.09, 0.4, 0.51], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.91015625, -1.8984375, -1.56640625], "student_probs": [0.13186147809028625, 0.36266180872917175, 0.505476713180542], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.09, 0.4, 0.51], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6d29676f85404e4e3ab5611c31ce7f6919d16619e161bd19dec4236413b1ab89:action", "state_id": "fc6a1750182d79ae1776dd68e1472c8178a87de094d417151960b12e94cdb6b8", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.36, 0.29, 0.35], "teacher_probs": [0.36, 0.29, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.26171875, -1.6796875, -1.4375], "student_probs": [0.40045103430747986, 0.2636500597000122, 0.33589890599250793], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.36, 0.29, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e3ff707e6d27562daddbe4e4b680f0e012c5c63e2d46fc983f80928180541508:action", "state_id": "3796655a65a34a4057679e710d371453f75ea15f6e522e8b33ea0f7ac3b0dc70", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.37, 0.25, 0.38], "teacher_probs": [0.37, 0.25, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.26171875, -1.41015625, -0.9951171875], "student_probs": [0.3156990110874176, 0.27214956283569336, 0.41215142607688904], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.37, 0.25, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3d7edbaf97ffc363f38cd2337693b4d5666104cafc394e68360536d7b225d901:action", "state_id": "d7725537f43d75926c81fbf5edb18871ab8e27178b75ee86ee549e8354c759c2", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.26, 0.24, 0.5], "teacher_probs": [0.26, 0.24, 0.5], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.3984375, -2.15625, -1.2109375], "student_probs": [0.18008585274219513, 0.22943532466888428, 0.590478777885437], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.26, 0.24, 0.5], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "30778036b73f372f2810864beaeaabcf1f61072c9b5054320588ebc59f2af908:action", "state_id": "e59a2eea2857fcb6a6b6fce0856cc618b0e5ef3e3834d48cf08b909f81587abd", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.2, 0.32, 0.48], "teacher_probs": [0.2, 0.32, 0.48], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.609375, -0.768798828125, -0.328125], "student_probs": [0.14453288912773132, 0.3349841833114624, 0.5204829573631287], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.2, 0.32, 0.48], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3d724a35a32696beefb042c1d80c26b3dfdc04b74714e9ff587cfa371fe9a13c:action", "state_id": "5b8b6a715f41ef4d385f78b7dd2cad25032a2feac7026f0ec399b6fc23a9eaf7", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.45, 0.55], "teacher_probs": [0.45, 0.55], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.0107421875, -0.8515625], "student_probs": [0.4602889120578766, 0.5397111177444458], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.45, 0.55], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1306b4decf07315794546797c07e63516825364e6bbf5811aa38b88f59e8941b:action", "state_id": "759753750a5ce88649bd43c8d0cd357f42966242465c7f39ff11a090996b5f3a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.16, 0.58, 0.26], "teacher_probs": [0.16, 0.58, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.21875, -0.296875, -0.03125], "student_probs": [0.14721311628818512, 0.3700937330722809, 0.4826931953430176], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.16, 0.58, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8aec15b1843571e22c7f67db6bf05cbdd4659c7d33187c86245c8a4f5cbd4ebb:action", "state_id": "17e6730c40a03ad8f52b8e0af0ba730346eddf5695a2fd912986f161073018bc", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.2, 0.8], "teacher_probs": [0.2, 0.8], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.61328125, -0.5048828125], "student_probs": [0.24816958606243134, 0.7518303990364075], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.2, 0.8], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c18d03ca50dde81c74889c819a6cd402028514da3c7711c2628d8f3b5b727171:action", "state_id": "ef39899ac5293db504e4c40a34ea68d9354720c3710d7bc4246d07f9cbb1ee70", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.31, 0.69], "teacher_probs": [0.31, 0.69], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.76953125, -1.53125], "student_probs": [0.4407099485397339, 0.5592900514602661], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.31, 0.69], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9db9663f6d672c694dc42faa419981207ba60507323bed37091a744eb5af129a:action", "state_id": "f07a9c168a2a5634e78946e97d5ab81909bf0b88ab0539a113619644f94b9e20", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.39, 0.48, 0.13], "teacher_probs": [0.39, 0.48, 0.13], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.586181640625, -1.14453125, -2.0390625], "student_probs": [0.5536950826644897, 0.31679806113243103, 0.12950678169727325], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.39, 0.48, 0.13], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cbcb94a7ca33b53b28344d1e4d258937fc5fcb1fe08de9b2c7a3d95f4b048a9c:action", "state_id": "e38a9978b92a81f7b518d3a8d5b11ff2776c0035c24f27a5c937ad8f2ddab76d", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.4, 0.52, 0.08], "teacher_probs": [0.4, 0.52, 0.08], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8486328125, -0.9453125, -2.20703125], "student_probs": [0.46191108226776123, 0.4193444848060608, 0.11874447762966156], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.4, 0.52, 0.08], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9ca534630fb997d1616866b3809bd26faa0135f85e39611c6ab29e7edc4d2d3e:action", "state_id": "190d0be73b50c991f941a327ef1bf2489901ba6bea6332b9e0fa9b1fb4502ca2", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.65, 0.21, 0.14], "teacher_probs": [0.65, 0.21, 0.14], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.57421875, -2.7109375, -2.796875], "student_probs": [0.6190735101699829, 0.19864222407341003, 0.18228434026241302], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.65, 0.21, 0.14], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "80e84d41c1a32105aa47e0f62fd948e56204de65423d89b8f39285c87b3991ea:action", "state_id": "f9f6ee6d8d424804cc4538cb80e1259f152a6614292895bbcf20341844bf379a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.33, 0.67], "teacher_probs": [0.33, 0.67], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.69140625, -2.29296875], "student_probs": [0.4016878008842468, 0.5983121991157532], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.33, 0.67], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e0654f71568fedc5250160f2a1b9990dc19c5bed5bf800635f49bc3844f971c9:action", "state_id": "5bdbd7dd2e448c892fac74c4f1d00eba527a555b3eeb0f9fefc2377454719be1", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.28, 0.23, 0.49], "teacher_probs": [0.28, 0.23, 0.49], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.5703125, -1.578125, -1.69140625], "student_probs": [0.3474431037902832, 0.34473928809165955, 0.3078175485134125], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.28, 0.23, 0.49], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0a8efc4e8d994333a2dac3ad32a84afa797e122479b6e605529e2a4f1c942d09:action", "state_id": "bd61343f7c6a6e8a67991d37c4ed1d30f152063a5163dd7cf6e4c667da6c8cb5", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.5, 0.5], "teacher_probs": [0.5, 0.5], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.11328125, -2.109375], "student_probs": [0.4990234375, 0.5009765625], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.5, 0.5], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "934ec509ed643a1b07718330b1d3e55dcb82d6d61e8d5285f2840f984bda8936:action", "state_id": "8c9a45be6351340ec851c2f732cecf7b96cfd73a8fa5221d290e6439917f4040", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.3, 0.38, 0.32], "teacher_probs": [0.3, 0.38, 0.32], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.857666015625, -0.14453125, -0.19921875], "student_probs": [0.201119527220726, 0.4103597402572632, 0.3885208070278168], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.3, 0.38, 0.32], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0ca04652855bf239a8eb0e592638209ee84e75f91d000dc59708e5d132d7c1cb:action", "state_id": "98e69fe92b50b25047139139b8c283e4760a02af30077cd32c56fc9f05d3ff51", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.25, 0.07, 0.68], "teacher_probs": [0.25, 0.07, 0.68], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.74609375, -1.79296875, -0.86328125], "student_probs": [0.22873368859291077, 0.21825921535491943, 0.5530071258544922], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.25, 0.07, 0.68], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1941285236e2e0041140a0e906599ce2fca651f884eb734d95a8566f29c722a1:action", "state_id": "6ae86bbde3f4b107de28e17c22e62aa173c0f5971f6ccaf99014eac2c093f7fd", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.15, 0.08, 0.77], "teacher_probs": [0.15, 0.08, 0.77], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.53515625, -2.58203125, -1.1484375], "student_probs": [0.35421109199523926, 0.12433978915214539, 0.5214491486549377], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.15, 0.08, 0.77], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7e7de9779f13c7036044673a857ac97fd77a9d515d87d1dc5bd5ddd5413c92a1:action", "state_id": "32b0ff3c79ba6c9efe44e3e8efe280640258cddb6ab37ee1c6c789a55c77dd19", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.6, 0.15, 0.25], "teacher_probs": [0.6, 0.15, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.046875, -1.068359375, -1.41015625], "student_probs": [0.37395259737968445, 0.3660041391849518, 0.260043203830719], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.6, 0.15, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ed611366d39a341540de872e36b8926d0093341c8e326f207a059c76dd278051:action", "state_id": "171a184e1d2f3daa5340ce1de33a0979763338ca47b896173e4b483bae8a7e8b", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.58, 0.42], "teacher_probs": [0.58, 0.42], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.34765625, -1.44140625], "student_probs": [0.5234203934669495, 0.4765796959400177], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.58, 0.42], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "be47c63b8f3f79528dbb43ea4ed71ddbb4fb163c4d607d5d0999eedd62423997:action", "state_id": "0f5b8eee86d7a4e8a7ce5f3a1421c748c4858f63b23bf52b42970ef8b7929c6d", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.46, 0.31, 0.23], "teacher_probs": [0.46, 0.31, 0.23], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.34765625, -1.8046875, -1.953125], "student_probs": [0.23782357573509216, 0.40932026505470276, 0.3528561294078827], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.46, 0.31, 0.23], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fa37692cbd84df2e66ef92dc8790ff5175c80906e292451bcf8c51b06d5b2011:action", "state_id": "37975e8a69916a7104bb1bf964560a54827210daf74e809aa3f37c2673e87412", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.74, 0.11, 0.15], "teacher_probs": [0.74, 0.11, 0.15], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.212890625, -1.62109375, -1.08984375], "student_probs": [0.3576817810535431, 0.23780250549316406, 0.40451571345329285], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.74, 0.11, 0.15], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f17c1af8b976879a4d6e2b2d680a3d058ef7631aaab08001ffb38adb28191a20:action", "state_id": "9c27bfa6a3b4c1183a411612fe8f0d82f918a7e984e32936b9ffc2a5454f2818", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.65, 0.16, 0.19], "teacher_probs": [0.65, 0.16, 0.19], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.255859375, -1.72265625, -0.8701171875], "student_probs": [0.322818785905838, 0.2024097740650177, 0.4747713506221771], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.65, 0.16, 0.19], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0a140dca29c980a84857cda5aba3c9a5b8c5d33c571c1c15efbb40598dd89b04:action", "state_id": "bb60de36c2241f6d7994152bae855008f2567b46f1dec6ab7bb419313e5b9f97", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.17, 0.83], "teacher_probs": [0.17, 0.83], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.56640625, -1.90234375], "student_probs": [0.5832033753395081, 0.4167966842651367], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.17, 0.83], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9adcadfd869a6ce30b2961dee5a69a2f1453fef75dc7616c85597ab6da8ca795:action", "state_id": "f690e8be2bc00f1dd9a3422bfea0922e4c3f486c28dbeb863e6cc4f05bb8159c", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.68, 0.32], "teacher_probs": [0.68, 0.32], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.34765625, -1.35546875], "student_probs": [0.5019530653953552, 0.4980468451976776], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.68, 0.32], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e0c37ab6c53915edaff382cbc8833909bc561525fa230526f3c898c280d90999:action", "state_id": "e320e7ff4864c0734ca5aabfd45592bbec7c5ef619852fe6d2a885d888de3fa3", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.71, 0.29], "teacher_probs": [0.71, 0.29], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.1484375, -1.38671875], "student_probs": [0.5592900514602661, 0.4407099485397339], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.71, 0.29], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "121d554597332f647b1ae4190391404cd5ce7bb560f5190a8a1114455c2775d1:action", "state_id": "50ab3d9d37fd48e778623dbba7dd40b3a5b9670ebd2e528fe082d644bd2f9830", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.28, 0.14, 0.58], "teacher_probs": [0.28, 0.14, 0.58], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.9521484375, -1.8203125, -0.60546875], "student_probs": [0.35284754633903503, 0.1480976790189743, 0.4990547299385071], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.28, 0.14, 0.58], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "37d331cb3269f9e330d27ee33a6e82774fd04a26be260658a947b43e22dd6af5:action", "state_id": "403d72e0f8e9e74ca6aad9683a87f1c5f4aaa1fc9967598254a832ed2cc2c7a3", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.15, 0.3, 0.55], "teacher_probs": [0.15, 0.3, 0.55], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.74609375, -0.880859375, -1.33203125], "student_probs": [0.20456112921237946, 0.4859478175640106, 0.3094910979270935], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.15, 0.3, 0.55], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fedf4286ba01156e66ebc251a00231a676e611019b4087bc97dfc0eaf454bc00:action", "state_id": "52a1b34f7dc0979b6bf519ccd7a86153998e58c86b3eb25dc33dd71e8fc075d9", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.24, 0.15, 0.61], "teacher_probs": [0.24, 0.15, 0.61], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.181640625, -1.298828125, -0.84130859375], "student_probs": [0.303505539894104, 0.2699434161186218, 0.4265509843826294], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.24, 0.15, 0.61], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5b2914c68c92a6c39a75df8ede0ae02565a80dd055edc740638264c4479e3ab2:action", "state_id": "f7597c546b159b2aeb41b195acd23b758ad6881ae5480f0e59eb61e0433c6679", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.33, 0.22, 0.45], "teacher_probs": [0.33, 0.22, 0.45], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.26171875, -1.3359375, -0.427734375], "student_probs": [0.23635391891002655, 0.21944719552993774, 0.5441988706588745], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.33, 0.22, 0.45], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "32e4d89a57dc9e50cea2283a089dc1e7662e43e8dcefd50850ab3414c00ae99c:action", "state_id": "1be3ec722d8a7a190a62b115220e81cf01398fd1f75ac7425c70829b5d8f9cd9", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.27, 0.23, 0.5], "teacher_probs": [0.27, 0.23, 0.5], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.08984375, -1.283203125, -0.4765625], "student_probs": [0.2724301218986511, 0.22453302145004272, 0.5030368566513062], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.27, 0.23, 0.5], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "147730e88c95c3ba1c8571f0813e9a0bd4e2c79c86e42f58f8738b31fecd9b15:action", "state_id": "3d33fc831372987396fcb9a7fdd85ab828521082bae7a9e2591be6a058e08956", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.29, 0.28, 0.43], "teacher_probs": [0.29, 0.28, 0.43], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.03515625, -1.96484375, -1.16015625], "student_probs": [0.22362765669822693, 0.23991744220256805, 0.5364548563957214], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.29, 0.28, 0.43], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b93371d0093972a67dd046df70e777dda985b94bf4bd68e95096cb4f657b3063:action", "state_id": "40eb167fe6458a9f9e09824fa50ef1efdd4f44068bc3d6f830558d160877e954", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.33, 0.5, 0.17], "teacher_probs": [0.33, 0.5, 0.17], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.46875, -1.32421875, -2.55859375], "student_probs": [0.4013216495513916, 0.463726282119751, 0.13495203852653503], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.33, 0.5, 0.17], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b777090a3a2574ddddc1e06298f9df71c94dc278cf96489e951e5768159df428:action", "state_id": "a9e97e5f4087a8198f22ff8a2789ec912d9e4ae5830bdc11556d0b434d3badc0", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.59, 0.15, 0.26], "teacher_probs": [0.59, 0.15, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.029296875, -0.91015625, -1.1640625], "student_probs": [0.3332834839820862, 0.37545326352119446, 0.29126331210136414], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.59, 0.15, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f7cc8b96ab2689d6a747b12baf9e87e43a05f7429a6ebf91a6017a8b83d25265:action", "state_id": "fb06a1754f9136213229fe5881baa01239ea799bea2bc1c9c8cd510416110482", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.68, 0.11, 0.21], "teacher_probs": [0.68, 0.11, 0.21], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.505859375, -0.8349609375, -0.8720703125], "student_probs": [0.4144345223903656, 0.2982146143913269, 0.2873508632183075], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.68, 0.11, 0.21], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4e96d4589eb4a1de4cd740e1e1cc32f1f11c05bff3c3326a143a712743ac6d68:action", "state_id": "b0168d6267841e07d78a10498f584c58dcb663a907f927abffa5976cd975409a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "west"], "gold_index": null, "teacher_raw_probs": [0.48, 0.52], "teacher_probs": [0.48, 0.52], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.41796875, -1.54296875], "student_probs": [0.531209409236908, 0.4687906503677368], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.48, 0.52], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e56d1e6a9339ffa2cb25f432d8f5db3f2c2c9e75ee3dbd49d7684500cfb69af7:action", "state_id": "9d69abcb8b3d5b254b4d35e5fccc48eba36ebd26894a37d58495b43e5063b484", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.47, 0.53], "teacher_probs": [0.47, 0.53], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.9853515625, -1.126953125], "student_probs": [0.5353413820266724, 0.46465864777565], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.47, 0.53], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5c25fed2d82a9dfbedc31be74f3e1fe14fe15e3ca90a207e1bbd30edfd813474:action", "state_id": "1d857e496ee2b4b9f6b0cf70dfd4eff8a42f9535c1dc76e3f1741f0e370de1ac", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.36, 0.27, 0.37], "teacher_probs": [0.36, 0.27, 0.37], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.359375, -1.51953125, -1.125], "student_probs": [0.3209109604358673, 0.2734195590019226, 0.4056694805622101], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.36, 0.27, 0.37], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "da0c56c6eb65a1f6988f33f706dfe1b89f8065c537a8c1f9d7f1abbcbd2bb336:action", "state_id": "af2f950aeaf98d88da09547cf6df15a34feb48a3eca01198eac9968cdc7568b2", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.29, 0.71], "teacher_probs": [0.29, 0.71], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.921875, -1.53125], "student_probs": [0.4035668671131134, 0.596433162689209], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.29, 0.71], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "acdd8bf5c744f90466debd59dcb2b9d137c2008643af5f1676cf6d972a69077c:action", "state_id": "2e66a54ef170f096c43f72c32a22fc09b9fc8ca556ad7ebe65bc1b5fc24af3ea", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.35, 0.32, 0.33], "teacher_probs": [0.35, 0.32, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.72265625, -1.4609375, -1.609375], "student_probs": [0.2924739420413971, 0.3799707889556885, 0.3275552988052368], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.35, 0.32, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "62c3394f8b056f146115de8d08ef1ca3e8ecf5df569b440706d0276b26037d10:action", "state_id": "fcfa64b8aab27146ed01f7b4e5b08cc4e3e2610725d618da76877065faeef05a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.37, 0.63], "teacher_probs": [0.37, 0.63], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.056640625, -1.62109375], "student_probs": [0.6374822854995728, 0.36251771450042725], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.37, 0.63], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a6d0f0c6b1b7506318188f19fc8117b2b82fbb851505a269eab9af6e26923ee7:action", "state_id": "c65a96d44ac13938e160a50e645bf0355239b658acf5b8041eda8ddb4d878c50", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.21, 0.59, 0.2], "teacher_probs": [0.21, 0.59, 0.2], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.6875, -1.2890625, -1.6796875], "student_probs": [0.28593170642852783, 0.42589402198791504, 0.2881743013858795], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.21, 0.59, 0.2], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6105348a7881a17135ee59ab1f5226eda1a1492568351719508e12e0493b0279:action", "state_id": "635367a6e4adbba27f73e6a7186ac9a6224c65c6f4efc9596997c6faad75ba57", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.24, 0.76], "teacher_probs": [0.24, 0.76], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.859375, -1.46875], "student_probs": [0.4035668671131134, 0.596433162689209], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.24, 0.76], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "be278c49acc7be4568eda1a0023e4042bd2a32f307b4a76407033314448e87a3:action", "state_id": "893148370307b3776c27d6bcf6b47ae97122fd6dc908061a54b702409fd3ac44", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.23, 0.59, 0.18], "teacher_probs": [0.23, 0.59, 0.18], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.67578125, -1.064453125, -1.6875], "student_probs": [0.261013001203537, 0.4810149371623993, 0.2579720914363861], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.23, 0.59, 0.18], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "87fafd5d706bbb599da7511abd061d0e36dc256531e48bc39bbe7dc3a2b819c9:action", "state_id": "b150982c6597ca39e3543304515d4626a1c7ba1e3dbf52d953508352333a0658", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.69, 0.13, 0.18], "teacher_probs": [0.69, 0.13, 0.18], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8583984375, -1.0458984375, -0.6708984375], "student_probs": [0.32946112751960754, 0.27313289046287537, 0.3974060118198395], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.69, 0.13, 0.18], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a0d94041968effeeef6172e418c66c7c849b53272a42540b48c841a711efe244:action", "state_id": "1c948a06b2913c305bc02d1d7a3f2257f061197560395894e2353d399afe7eef", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.59, 0.2, 0.21], "teacher_probs": [0.59, 0.2, 0.21], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.774658203125, -1.30859375, -0.7501220703125], "student_probs": [0.38297557830810547, 0.22453589737415314, 0.3924885392189026], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.59, 0.2, 0.21], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "907d4475b1d1e8a9a295502928d83698a63bd56e368323cc14c860d85c722c74:action", "state_id": "29ecc972c1572fec86d5bce647c6a8a06e90cdece61a90fe141e0f484117482e", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.3, 0.42, 0.28], "teacher_probs": [0.3, 0.42, 0.28], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.03125, -0.9677734375, -1.25], "student_probs": [0.34854656457901, 0.37138840556144714, 0.28006505966186523], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.3, 0.42, 0.28], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9c1871db0ed88d50592f9272d4212431c82839568061273efb8905f68b92055b:action", "state_id": "c64dcd23601f93f48dd39f83ea390c9c38ca5c741307955b50658764dcc42e7a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.39, 0.61], "teacher_probs": [0.39, 0.61], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.12890625, -0.5234375], "student_probs": [0.3530935049057007, 0.6469064354896545], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.39, 0.61], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cbda5e7fa603305c474ce4351caba47cfa6ff2ea10802d3d026dce48579ac2e4:action", "state_id": "2657e991d3e76cdcdc050aa9ed22fa8bc40273e4f7df980b8236d3c0ecfaaa15", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.35, 0.27, 0.38], "teacher_probs": [0.35, 0.27, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.177734375, -2.05078125, -0.8408203125], "student_probs": [0.35482436418533325, 0.14820197224617004, 0.4969736337661743], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.35, 0.27, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "74880447dbcb3341d69bfec53d15a03c503c86a1bf5232abaf7838cba71cedaa:action", "state_id": "057e31502d6b19237be68a260ac909a6d26e1fe41e84479aba091b27976f4763", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.5, 0.5], "teacher_probs": [0.5, 0.5], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.8125, -0.77783203125], "student_probs": [0.2621801495552063, 0.7378199100494385], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.5, 0.5], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "82c12e3e324a571a74cca592372f0a540db138a44ca3a1c7a80ffa31df75e21b:action", "state_id": "9c0874e90469b8259540253fee2085d42b3730084c5c28c663a0de1144f308f1", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.26, 0.34, 0.4], "teacher_probs": [0.26, 0.34, 0.4], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.66015625, -1.083984375, -1.578125], "student_probs": [0.2587520182132721, 0.4603753089904785, 0.280872642993927], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.26, 0.34, 0.4], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "18f851302e1249d844f40a3b6f3b1a8c9b694a13e76fc8a1637c8dbe781d5a7f:action", "state_id": "cb82502d69ea0b1b6fb6d263891cec72fd10e3d87fd7a5bbda5712bc068bc580", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.2, 0.5, 0.3], "teacher_probs": [0.2, 0.5, 0.3], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.6171875, -1.080078125, -1.62890625], "student_probs": [0.2703138589859009, 0.462521493434906, 0.26716458797454834], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.2, 0.5, 0.3], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a285c9cb1053a45be48a2816b467f2fb3fc96967cbb34f7a01fad8fbc771f294:action", "state_id": "d894b0600ef470323a124b00a9c09f94e49c5b787eea473ad12b5c2f93e436ac", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.2, 0.53, 0.27], "teacher_probs": [0.2, 0.53, 0.27], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.76953125, -0.77001953125, -1.81640625], "student_probs": [0.21407951414585114, 0.5816444158554077, 0.20427611470222473], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.2, 0.53, 0.27], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "751ed9a0789ba8ff65777ec264ab4c67f3e75950f0971a0c0ca3b9140705be72:action", "state_id": "1d6da4f2ceb943934d1c7149e0c2a13df4751e14a43c858604a6b16198ce5c46", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.18, 0.41, 0.41], "teacher_probs": [0.18, 0.41, 0.41], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.2109375, -0.306640625, -1.083984375], "student_probs": [0.2171289473772049, 0.5363507270812988, 0.24652035534381866], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.18, 0.41, 0.41], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8a6c5b8a158e68fa534be1f9e0344727f6d86d7da74b81afc4b87b63af29fb9f:action", "state_id": "24428bbbfa2640c0ef4c3f98185f5847b0c939c702b3330c7ce06a35a1c3c386", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.2, 0.36, 0.44], "teacher_probs": [0.2, 0.36, 0.44], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.546875, -0.25390625, -0.4609375], "student_probs": [0.29153338074684143, 0.3907715976238251, 0.3176950514316559], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.2, 0.36, 0.44], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "85b98deb717b2849c44142f70d7a4ff1d4255f36f135fc15d39edaf5e5b1aadd:action", "state_id": "280ac21d8c8c5c90aa5df99ffc31eae98fc16d98d88fe8dd67b945f105f85668", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.48, 0.52], "teacher_probs": [0.48, 0.52], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.9365234375, -0.729736328125], "student_probs": [0.44848665595054626, 0.5515133738517761], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.48, 0.52], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9d0ecb29ab31e89fa21b5a3683ecac92789ba21d108b57db65bccaaa7efd59d8:action", "state_id": "7a7af1cf5361291844b1e99408f3bb5dacfdf0f0f0a395af5bc4eb2aa503ab11", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.29, 0.43, 0.28], "teacher_probs": [0.29, 0.43, 0.28], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.03125, 0.578125, 0.21875], "student_probs": [0.2541893422603607, 0.4391998052597046, 0.3066108822822571], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.29, 0.43, 0.28], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0bf458ebbda937c60db9c584503e215b5261edb0bdde2738878d1aba4d343bda:action", "state_id": "dc8728ad30ec0bced275e3adddb2232b1ff08da596b3744b307da4dd18197d53", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.68, 0.23, 0.08], "teacher_probs": null, "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": null, "student_logits": [-0.69580078125, -1.19921875, -1.326171875], "student_probs": [0.4679774343967438, 0.28287413716316223, 0.24914845824241638], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": "Rounded target has no identity probability-simplex representative", "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": null, "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7e58fd03b85d98925bd5474139ae9e79810ccff8dda174e34cbdc57ce304f434:action", "state_id": "687aa1f30f805e0b1a166393961ab99e064416b363618738d621b7744c950688", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.34, 0.54, 0.12], "teacher_probs": [0.34, 0.54, 0.12], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.71142578125, -0.4951171875, -0.9228515625], "student_probs": [0.32777053117752075, 0.4069223403930664, 0.26530709862709045], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.34, 0.54, 0.12], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a1c26323fb3429dfc5a51b91c79df95cdfb80bbc287277c9835b56a99d9cf03f:action", "state_id": "2d20983affddeccaea9d24acc2bd32e506f9747a1caafe2281b0b0f56e29be27", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.75, 0.11, 0.14], "teacher_probs": [0.75, 0.11, 0.14], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.494140625, -0.482421875, -0.451171875], "student_probs": [0.7803432941436768, 0.1081124022603035, 0.11154425889253616], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.75, 0.11, 0.14], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6a1cd23727139daa0187f897277ab0e0dc54005fabf1cf7a773a70bf8bd5be9e:action", "state_id": "28865b9658718e77377887bd4d88870fdf3a3c66f6ea11fcf3907f91d42bc07d", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.75, 0.19], "teacher_probs": [0.06, 0.75, 0.19], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.875, 1.740234375, -0.0390625], "student_probs": [0.05890185013413429, 0.8052130341529846, 0.13588514924049377], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.06, 0.75, 0.19], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "187636a56982cc0a3c7c08b0a3472f27f0135c3b3da20c05208ab0d26622bfcc:action", "state_id": "6d41ecedc038569196aed370d3fa7f20a352356dcae92e466a8b479c58ab1df7", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.55, 0.41], "teacher_probs": [0.04, 0.55, 0.41], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4912109375, 1.93359375, 0.90234375], "student_probs": [0.06124011054635048, 0.6920145750045776, 0.24674539268016815], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.04, 0.55, 0.41], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1dc60275754c86c7dc32afc3081531ecbbd11ea45b56145ad064e8138ba7e38f:action", "state_id": "d7ad8484efba18eb569e34adee7382d92e965bf0f5aa6c86094625cfd020f5d3", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.0, 0.04, 0.96], "teacher_probs": [0.0, 0.04, 0.96], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.294921875, 0.296875, 4.1640625], "student_probs": [0.011209889315068722, 0.020258881151676178, 0.9685312509536743], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.0, 0.04, 0.96], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "93c55ae01724fb47ae209b0320d55a10d995c9f0916a481c4b3ecfa79a84fbf1:action", "state_id": "35da4b37d898b07669b41fe07921dbb072610232cc7c55c05192cae31a23a675", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.36, 0.59], "teacher_probs": [0.05, 0.36, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.38671875, -0.19140625, 0.05078125], "student_probs": [0.11744328588247299, 0.3881019651889801, 0.4944547116756439], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.05, 0.36, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0c7c84da13768c35e0ddb993b05e1e3a9b0de349653a354a368d201dee9e43ec:action", "state_id": "8fdc8644fa01b4e19af8360b541ff1679cee53beef20d56051177d910d6a9b54", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.92, 0.01, 0.07], "teacher_probs": [0.92, 0.01, 0.07], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [3.009765625, -1.45703125, 0.80078125], "student_probs": [0.8918250203132629, 0.010241756215691566, 0.09793319553136826], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.92, 0.01, 0.07], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ff963fc724fd47c644c64970454f34e596e3ac65eb51dec714fe9854bf9967a1:action", "state_id": "976acb34ea5463f76c343bc659adbe2b6978bd89c93a61acfd674724dbfe09cd", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.61, 0.33], "teacher_probs": [0.06, 0.61, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.75390625, 0.15234375, -0.4404296875], "student_probs": [0.0873599573969841, 0.5877413749694824, 0.32489874958992004], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.06, 0.61, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f5004680ab96105932aaa744378f5b5a490a59e2c5e3bda7ea1c2ac3b5728295:action", "state_id": "0fcae231aa2a1fef0218eb9a89182b95d0917fc67c26483b9956f83b7b778442", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.1, 0.58, 0.32], "teacher_probs": [0.1, 0.58, 0.32], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.48046875, 0.37890625, -0.171875], "student_probs": [0.08992248773574829, 0.5772774815559387, 0.332800030708313], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.1, 0.58, 0.32], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "711d6f8f01a168c7d8e16cf2f67256795a24bf085d7100d206f4da8cc2850b3a:action", "state_id": "ad53bfd682c8d39a0c7d8c297aa7a8d52a1cb72972604628995536d65ff472af", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.08, 0.5, 0.42], "teacher_probs": [0.08, 0.5, 0.42], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.4453125, 0.16015625, 0.3203125], "student_probs": [0.08456361293792725, 0.42114317417144775, 0.49429330229759216], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.08, 0.5, 0.42], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ebfe9bb3911d957ce66420ff974ff5be5e3e07a2c485ebe6509487418746afca:action", "state_id": "4efd532ca67c776526c0bb3a1ed896c480503848db36e60bc3ca18f676a37182", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.37, 0.58], "teacher_probs": [0.05, 0.37, 0.58], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.32421875, 0.21484375, 1.1484375], "student_probs": [0.05709681659936905, 0.266083687543869, 0.6768195033073425], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.05, 0.37, 0.58], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fe47f917e44915959650fe6a5443787a96a141d80c869bd4f9892e395f317c76:action", "state_id": "b7e8ced96070a3e9e079c8ffab966e727b04d5b1d40423038536c1a931412534", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.04, 0.87], "teacher_probs": [0.09, 0.04, 0.87], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.375, -0.576171875, 2.671875], "student_probs": [0.08826632052659988, 0.03409622609615326, 0.8776374459266663], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.09, 0.04, 0.87], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "65953e699660bbc221730bedff8286d886cb778904d7708b10847846a94d129a:action", "state_id": "6fd0cf2b21fe60217a71b4cc7940179eab8f752b0d06063cc2052a7236ae358a", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.08, 0.03, 0.89], "teacher_probs": [0.08, 0.03, 0.89], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.515625, -0.80126953125, 2.822265625], "student_probs": [0.08842825144529343, 0.023695778101682663, 0.8878759741783142], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.08, 0.03, 0.89], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "44508dae09a4c81afb692d2c69508c0afa0ec965d2ea21786643b58f7eb5698e:action", "state_id": "3c8cb6d03a39552fb5de2c8e3934aba9ef75ab62f2f13f80c4655a4c8238966c", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.02, 0.92], "teacher_probs": [0.06, 0.02, 0.92], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.2109375, -0.400390625, 3.2099609375], "student_probs": [0.046277955174446106, 0.025111792609095573, 0.9286102652549744], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.06, 0.02, 0.92], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7cc0829deaa737fad12e9ec7456f1d7d5a5e7a784aafda0742e1cf7e2f93bca6:action", "state_id": "a7d2e9be81991cf0711906cbd18ce55711328aadcd3dd05d79d499a508d2f73a", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.01, 0.96], "teacher_probs": [0.03, 0.01, 0.96], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.296875, -0.578125, 4.109375], "student_probs": [0.011945093981921673, 0.009016630239784718, 0.9790382385253906], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.03, 0.01, 0.96], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c533015f6d5db1230b7005df44d20cb7d7df6546852f8efd1aeb30a35d6f6218:action", "state_id": "b4c979289c9f62aed03f1e78e65c1b008c0264b5c74e70865fac5ed5a4994e6e", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.58, 0.38, 0.04], "teacher_probs": [0.58, 0.38, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.04296875, -2.2734375, -1.8984375], "student_probs": [0.3390222191810608, 0.26923832297325134, 0.3917394280433655], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.58, 0.38, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "538dd39369ac10a5b8815320da3535e675a440378740fcd7329cc6ba53abb197:action", "state_id": "9f2e7508a60cba421e7ddf2d53a02bbaf8820d702fa6c32195009bccbf385a40", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.5, 0.49, 0.01], "teacher_probs": [0.5, 0.49, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.3046875, -1.47265625, -4.0], "student_probs": [0.5227660536766052, 0.44193610548973083, 0.035297833383083344], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.5, 0.49, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7055874a6fe3c3cdeeaebd5fbc9c50f3a6e8bfc4cd7bd8cf15dfaafa81f49d33:action", "state_id": "a22abb7eaf42d252f6cd47f197855d08c9a13ccea182e87a7e922fe4a29df084", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.67, 0.17, 0.16], "teacher_probs": [0.67, 0.17, 0.16], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.83203125, -0.927734375, -0.9609375], "student_probs": [0.7470768094062805, 0.1285608559846878, 0.12436231970787048], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.67, 0.17, 0.16], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "60644f41e885c722f07f02462b62f6f1ff38172d97b4ddfa9a87eae7e693ac60:action", "state_id": "8e23b13c7826edad5ca436b1f41937ef04c5f93d462d938e135f22b8a01fb158", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.25, 0.63, 0.12], "teacher_probs": [0.25, 0.63, 0.12], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8525390625, 0.15234375, -1.66015625], "student_probs": [0.2393772453069687, 0.6538797616958618, 0.10674294084310532], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.25, 0.63, 0.12], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2a186a3bb8fd988e023c9088ff5353857b47d14173ac51be42855cc1c0027860:action", "state_id": "3df6f41d60d018e4915c37ee83cd6fc47be954d63c469c778323142c0332895f", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.37, 0.56, 0.07], "teacher_probs": [0.37, 0.56, 0.07], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.32421875, 0.65234375, -1.90625], "student_probs": [0.2590089738368988, 0.6877498626708984, 0.053241144865751266], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.37, 0.56, 0.07], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "db12abfbd2d302d495427a5b7cc4761f4d443a0c08bca46d17fea9e50a5107ed:action", "state_id": "45b960955dab1894bb651921344ffdfaabeac58efa284c56e849368918c46729", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.91, 0.06, 0.03], "teacher_probs": [0.91, 0.06, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.310546875, -0.4697265625, -1.33203125], "student_probs": [0.9189434051513672, 0.05699428915977478, 0.024062301963567734], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.91, 0.06, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "06109d09565c43961bdf3b0fc9a17f33f124059beb8d76842b0e96a19a1c33c0:action", "state_id": "d0f561d2d68b4fb77626e2228e7cc3d6a34ee3e8a5645be4ea1801038046a679", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.9, 0.08, 0.02], "teacher_probs": [0.9, 0.08, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.431640625, -0.01171875, -1.47265625], "student_probs": [0.9033231139183044, 0.07847035676240921, 0.018206587061285973], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.9, 0.08, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "06bcae81211ea3e2d4d0dcf19754ba5c618f4047757d000f6742f1495ba53c34:action", "state_id": "4ea6b13e3252f74b25611befb42c9d4be60e66f13891cefb95ce9fd98a4ecb5e", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.93, 0.06, 0.01], "teacher_probs": [0.93, 0.06, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.412109375, -0.103515625, -1.97265625], "student_probs": [0.9146803021430969, 0.07391750067472458, 0.011402230709791183], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.93, 0.06, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "20f5422c079ae0744411a2a162d7099a6834ce3fdf4a4b5e0a07edce7387ad33:action", "state_id": "659e860e5e4a8b0bfd09c8353cf52cc5bbb344540673e3998f35f96867675049", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.95, 0.04, 0.01], "teacher_probs": [0.95, 0.04, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.29296875, -0.28515625, -2.23046875], "student_probs": [0.9201597571372986, 0.06985504925251007, 0.009985257871448994], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.95, 0.04, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4706df15f4cc6027e47ea88ecea645ccb4e1f76fd0a43edd892105a853cf7230:action", "state_id": "46df381ee65cc98beee090260d5a9ab99276e8ea100265c6fdaf5cfcd472d4b1", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.99, 0.01, 0.0], "teacher_probs": [0.99, 0.01, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [3.779296875, -0.125, -2.04296875], "student_probs": [0.9774062633514404, 0.019699741154909134, 0.0028939915355294943], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.99, 0.01, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "00f95d897c8c054e718ce3db0a30da9f2decd8d0e3f51ac7c5f7ebc4a7d5560e:action", "state_id": "c943323d0f2a712229855f4872af3506fb1c3ddcbd0bfa9dc473eea383866f12", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.79, 0.19, 0.02], "teacher_probs": [0.79, 0.19, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.0, -0.0546875, -2.1640625], "student_probs": [0.7191373109817505, 0.2504764199256897, 0.030386239290237427], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.79, 0.19, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1a1684ca1b938022e588f2db29c79db0dade6bdf127d9c947c425ed1fbe76ab6:action", "state_id": "03aa3064e6ba696ba7d25ca5d51053ae32210d3ec9d091f44f7c64e8310c69c8", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.73, 0.25, 0.02], "teacher_probs": [0.73, 0.25, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.484375, 0.40625, -1.9375], "student_probs": [0.7283936142921448, 0.24782343208789825, 0.023782894015312195], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.73, 0.25, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "783902ec47316bba3618f7543dc98132b12839e4ab8b48a5ba67c57ad9ff3937:action", "state_id": "0c3e0c07e774cbbec0196578dc4e153f0a81ae1cfdce8ff187f9429bdf855cec", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.61, 0.36, 0.03], "teacher_probs": [0.61, 0.36, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.43359375, 0.56640625, -1.65625], "student_probs": [0.682295560836792, 0.2866538465023041, 0.031050631776452065], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.61, 0.36, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6fafa97190c75ce92907c37ee715d582fb0f071c942cee10f98455e9f177d253:action", "state_id": "354e672c361bb07f03d570187358da0c408d7d6ea09a45980bebe2936537e058", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.43, 0.53, 0.04], "teacher_probs": [0.43, 0.53, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.30859375, 1.294921875, -1.23046875], "student_probs": [0.256676584482193, 0.6882451176643372, 0.05507822334766388], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.43, 0.53, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "12c11ac0a1851ff94e13ced6f2989f96ac9594df317b3585169c46cbe53fe7cc:action", "state_id": "cd5788a923d73eb724847598c53dd80e8418ce43529a56d4b1474073c5eb8c8e", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.9299999999999999, 0.04], "teacher_probs": [0.03, 0.9299999999999999, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.078125, 1.13671875, -1.984375], "student_probs": [0.037040214985609055, 0.9222791194915771, 0.04068071395158768], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.03, 0.9299999999999999, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f2cfc8337e545a1fa8a98dd821269cb8e083dcea0206f65d4a508298068fc300:action", "state_id": "4d8e0f9dbc0574ec40986c55f7548a58f02b49d5d89905cf6d46fb23a3f9b07d", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.98, 0.01], "teacher_probs": [0.01, 0.98, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.65625, 2.193359375, -1.5703125], "student_probs": [0.02038135752081871, 0.9574083685874939, 0.022210344672203064], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.01, 0.98, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d8a4ff14bb614210e9f60b843acc064cf8587536e3945c3c796f5102798dbc4d:action", "state_id": "d5f977b2caac558a63c4212386b498407179c0f7ccbb40e5c8214ed528fdae64", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.0, 1.0, 0.0], "teacher_probs": [0.0, 1.0, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.23828125, 4.4609375, -0.8154296875], "student_probs": [0.0033204907085746527, 0.9916114211082458, 0.005068090278655291], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.0, 1.0, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "03a417af26b4ff2fb5458f78330eeb0806c054b26cbf9fa913e8b8f695cb7590:action", "state_id": "20dfe6f5b23bde3ed03df8ca874ee110f01ec0d13fdcfe2d99c1ca7e7d2c1264", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.43, 0.55], "teacher_probs": [0.02, 0.43, 0.55], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.59375, 0.30078125, 2.423828125], "student_probs": [0.015818830579519272, 0.10518621653318405, 0.8789949417114258], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.02, 0.43, 0.55], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8fde7751a88974e4684987942fde86aebeb3519f7c180c62bb60fa66d967e063:action", "state_id": "cd4b14c69c2c9641f0f2b235edd7621c6945156a2fc4fd36ac089730c8faebd8", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.93, 0.01, 0.05], "teacher_probs": null, "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": null, "student_logits": [4.642578125, 0.806640625, 1.6591796875], "student_probs": [0.9326604008674622, 0.020127834752202034, 0.047211747616529465], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": "Rounded target has no identity probability-simplex representative", "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": null, "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "32aada99fdcc866259b12e721e9fcc44893e95d587883a68c54982b2b50ae119:action", "state_id": "968515504effc10e6b30781030d0c25a3a440d58ff952877d1f36bd49c8b5c2b", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.35, 0.46, 0.19], "teacher_probs": [0.35, 0.46, 0.19], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.7088623046875, -0.326171875, -1.5390625], "student_probs": [0.3445678651332855, 0.5052136778831482, 0.15021848678588867], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.35, 0.46, 0.19], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "682ae21b32376787d358c24a6eaf1169bb9edea79e4dc166e2a7fe9b8268e984:action", "state_id": "a9f25dfd5efdbd1787a4d2bafd898a2fecd14d5fee9c9695da627336a3da2e39", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.65, 0.09, 0.26], "teacher_probs": [0.65, 0.09, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.671875, -2.1953125, -1.5390625], "student_probs": [0.36569535732269287, 0.21666733920574188, 0.4176372289657593], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.65, 0.09, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "107139c821975e2d002026f701aea24c6755eb0d62bf7442b06c08e1d5a6aeda:action", "state_id": "bf7f2ed904da0c63ceb963ad0be44cd64a771dd5c9d7775c0c5049f33b8f94a8", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.56, 0.03, 0.41], "teacher_probs": [0.56, 0.03, 0.41], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.53369140625, -3.34765625, -0.791015625], "student_probs": [0.5455286502838135, 0.03271358832716942, 0.4217577576637268], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.56, 0.03, 0.41], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "11cbc03aa902843368cf17b0f0f841773e6924b920dd967f20859db0c4855b0c:action", "state_id": "08cb978e69a55965bd0c070236bd80523bf05dc3b9501e61619fef3bdd6e7418", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.16, 0.6, 0.24], "teacher_probs": [0.16, 0.6, 0.24], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.1796875, 0.078125, -1.23046875], "student_probs": [0.07606969773769379, 0.7273897528648376, 0.19654051959514618], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.16, 0.6, 0.24], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "72f00737ffc434f9ed0a9b9356bb683795dedaf7bbc296a6043ac638ef38210d:action", "state_id": "912f0d45a1112013c0a880bcf5fc90a9dd6721e6bb785918e6de420364946731", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.12, 0.64, 0.24], "teacher_probs": [0.12, 0.64, 0.24], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.53125, 0.66796875, -0.4248046875], "student_probs": [0.07667796313762665, 0.6914792060852051, 0.23184281587600708], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.12, 0.64, 0.24], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "26ccb1a1b2c524a8599263f7ce3e3c60f47a62d613619bd1529718f367a702f1:action", "state_id": "b45beef2a45473ce87c50f656790ccde917a4625912b67eaafbb003bf661266f", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.61, 0.34], "teacher_probs": [0.05, 0.61, 0.34], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.35546875, 0.4921875, -0.22265625], "student_probs": [0.09570013731718063, 0.6072107553482056, 0.2970891296863556], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.05, 0.61, 0.34], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "dde3cdc92db2f8067244ea9a3e10e94e7dc7e8d33f9494446486eae2e8efe86f:action", "state_id": "29be622b7c96a638fad90d096348edc56bc64cb9f814edd99711fe58336a8472", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.63, 0.33], "teacher_probs": [0.04, 0.63, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.359375, 0.40625, 0.08203125], "student_probs": [0.09031905233860016, 0.5279351472854614, 0.38174572587013245], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.04, 0.63, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "93b3bc111b44c4a99cffb7c68b9e0a50125b2e9a9c4ad62f5812b89128dc7200:action", "state_id": "dca1534538e910f3b412c914e678c8e56912542510e532b55e29e2730a93c1b0", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.61, 0.36], "teacher_probs": [0.03, 0.61, 0.36], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.275390625, 0.38671875, 0.4140625], "student_probs": [0.08556564152240753, 0.4509665369987488, 0.4634678065776825], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.03, 0.61, 0.36], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "447a144a40257bdc271dd43c3b85d38d726a2eddb99c7be30634858530925089:action", "state_id": "94afcbc6fda82da67fa2b648b8c9504b33c3588903626c2a7b09ce83d43a375f", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.1, 0.66, 0.24], "teacher_probs": [0.1, 0.66, 0.24], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.09765625, 0.40234375, 0.48046875], "student_probs": [0.096828393638134, 0.43395471572875977, 0.4692169427871704], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.1, 0.66, 0.24], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4af15dda8a23f42c8546f886acee15878b85b8e9cac8a6ed38cb91d7b8b21905:action", "state_id": "2c5d8187793efed64254644c32068a0097a10d1cce3d11953438a224778cfec9", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.6799999999999999, 0.25], "teacher_probs": [0.07, 0.6799999999999999, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.205078125, 0.515625, 0.28515625], "student_probs": [0.09068985283374786, 0.5068162679672241, 0.40249383449554443], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.07, 0.6799999999999999, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e08f947dff4128182c9cd2b55b2b519eb7770fe763dd0f161fb3035bb25e3c87:action", "state_id": "32570c76e20bc2a24a41ed4f4575417b01178a760578a75bf42f9d666954dca5", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.5599999999999999, 0.4], "teacher_probs": [0.04, 0.5599999999999999, 0.4], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.232421875, 0.32421875, -0.13671875], "student_probs": [0.11449315398931503, 0.5430251359939575, 0.34248167276382446], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.04, 0.5599999999999999, 0.4], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "db25f986b6d1849127d13960bac3536ba9ce3c87f23ee575f21ed4e47d1586c3:action", "state_id": "5d404fe24823be58cbad82381e566e3616beef991ba3bd8d79bdf621a18ffabb", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.59, 0.39], "teacher_probs": [0.02, 0.59, 0.39], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.51171875, -0.0546875, -0.921875], "student_probs": [0.14090655744075775, 0.604939341545105, 0.2541540861129761], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.02, 0.59, 0.39], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6cc7792bc7d51635aea3be427ee85dbcb443d31fe6d6a31b9afa1a7bfca76cd1:action", "state_id": "48065dcf6d78a015574f022049466fe14161121c1bc9d9dffaa51746e6395e04", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.61, 0.35], "teacher_probs": [0.04, 0.61, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.4609375, 0.19140625, -0.80517578125], "student_probs": [0.1227625384926796, 0.6407219767570496, 0.23651547729969025], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.04, 0.61, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d902c3f52b799f0ac86174b7d4bcb2e3060e98b1e8e8274adae875aaba33fce5:action", "state_id": "04e0e41d595be4cad15a12b316eab38843dc75a14b2a689219970e8593d45c64", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.07, 0.91], "teacher_probs": [0.02, 0.07, 0.91], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.0185546875, -1.259765625, 0.23046875], "student_probs": [0.18965931236743927, 0.14901074767112732, 0.6613299250602722], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.02, 0.07, 0.91], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5265dcc276f8cf63314f06d1836bc18de9f549ecb79510a4aea2af9d94197ee9:action", "state_id": "fde3ac67d8a733efb730a2aafa87ead8cb9ba12973676018f4e20776df64b9e3", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.0, 0.99], "teacher_probs": [0.01, 0.0, 0.99], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.001953125, -3.4140625, 2.30859375], "student_probs": [0.035100363194942474, 0.0031459066085517406, 0.9617536664009094], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.01, 0.0, 0.99], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d16f99dee73ef00422df263ff1e6996e10e19e234de49605dd43fb8203b9b474:action", "state_id": "29755cbdf7165310d807e89a3d822f0da9fa36a82cee896605cfabdf4345898f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.78515625, -6.5, -8.015625, -2.55078125], "student_probs": [0.9950957298278809, 9.233635501004755e-05, 2.0283607227611355e-05, 0.004791777580976486], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5986e0a9989f971b3c212e721c25f590136049e099733598b9387654f3f73369:action", "state_id": "377b266e00a81127b5def44ac6ef3adb5f2e6a7216ebce7f55d4a343b852206a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.25390625, -5.31640625, -7.578125, -2.125], "student_probs": [0.9870613217353821, 0.0005088610341772437, 5.3008705435786396e-05, 0.012376826256513596], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d5071d33c47b8cc89a95a494b6b2295654639bcbaa9c79c0c9eb663cbc47e106:action", "state_id": "65a936c7b6b08902c7c44fea41958dbaaaf074b0c366f00a9260844f5d0396dc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.078125, -3.546875, -6.1796875, -1.474609375], "student_probs": [0.968511700630188, 0.003492998657748103, 0.0002510628546588123, 0.02774418331682682], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "749906f8fb0b21c825d49f513b57e8cb8aea0a974c4e2f5aa2ba65a569d00c55:action", "state_id": "86700703a3b65737bdbae9f3caa0bcd3ccfae2dd473fc6965a4ddc54f53ebffd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.59375, -3.01953125, -4.05859375, -1.0234375], "student_probs": [0.9204404354095459, 0.00913004856556654, 0.0032300851307809353, 0.06719943135976791], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "84eaa62c13b89d929a9c60f3d7954f1f777444e0802a5ad4761e6a71b821a564:action", "state_id": "15e9da7c6739351832ad42f621c2c88058ddd8352ff5ac43bb1336f73d8e6d6c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.51953125, -1.8203125, -2.734375, -0.181640625], "student_probs": [0.8116087913513184, 0.028765439987182617, 0.011531843803822994, 0.14809390902519226], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c8b2fccd158156240e96fd62d21f8e1dccd7488123687b24772b403310d5c22b:action", "state_id": "f73ae226db8d850b3c5cbac9c72ec3296244d0366af821b407ff71104c751c86", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.7265625, -1.546875, -3.08984375, 0.01171875], "student_probs": [0.619489312171936, 0.06378116458654404, 0.013632943853735924, 0.30309662222862244], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cf628574a6f73a78962ef873ecad4041e0edfd9ee59f5bf22f3fd1b0c511b552:action", "state_id": "d932b968f1d1701218adda24036902659f86f30b5551f6534b14dee2e73b1ce8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.39453125, -0.96484375, -3.453125, 0.3046875], "student_probs": [0.4561575651168823, 0.11715095490217209, 0.009729689918458462, 0.41696175932884216], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c75bb5cc5c172888cbf2b34a69a4459e4f62f735c426822d256770177abf3a7a:action", "state_id": "5f14246a0aab6e7e7a4aa2245678ef7cfa278d03d3469ceeb5591d2cba017a0a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.04296875, -0.76171875, -3.078125, 0.359375], "student_probs": [0.34922006726264954, 0.15618085861206055, 0.015403712168335915, 0.47919541597366333], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ba33c018d6bcaf3a55bad63874a76b2c6a84e784e94c18405461dc56546b3070:action", "state_id": "0335104574f880a275b3441f687103fa53adfe5172f37ed65a13c9a83a2d8e42", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0859375, -0.841796875, -3.05078125, 0.3046875], "student_probs": [0.37266242504119873, 0.14736884832382202, 0.016182884573936462, 0.463785856962204], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b48405eb390a72dbcb81d6bed46de49132c54b32783a89ff894b814d127d9d4f:action", "state_id": "e365a8df009ab79e75657abf31c2961e77fb9c2983bf69cee07cfb6ef3d20091", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.56640625, -6.3671875, -7.8203125, -2.201171875], "student_probs": [0.9914107322692871, 0.00013075044262222946, 3.057447247556411e-05, 0.008427927270531654], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cc7b029e028905008b1bbb62cb76f82690ff50f9d47a4d38f003c13cfb7700c9:action", "state_id": "9f7dc7a1a327a0254e3ce21e955aebcf08a434500b41b1bac4bc4d0c47deb5f2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.16796875, -3.24609375, -5.9140625, -0.9375], "student_probs": [0.9527747631072998, 0.004243192728608847, 0.00029444805113598704, 0.042687658220529556], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "28d2bca0f764f47f90ccdd9fa29f082bbe2eec1840b2292727fe6e283252ae91:action", "state_id": "09f460fd9d51ccc1394dd221e5d7f7fac47304d7b92784482b43d9388d95e20f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.1796875, -3.4296875, -6.171875, -1.37890625], "student_probs": [0.9686372876167297, 0.003548465436324477, 0.0002286249800818041, 0.027585672214627266], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "86e1854b6e57f60e8f672b9285baff0b851bb56100bc2e42f5bab0a22488b657:action", "state_id": "6d2cf967ecc1196005c8c0ba7e7b05d2c9a0be596ce5af669738b187e824901b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.8046875, -2.158203125, -5.125, -0.19921875], "student_probs": [0.9461225271224976, 0.006615936756134033, 0.00034050841350108385, 0.0469210222363472], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0127d22bbd67f901f001959fcee7c8fa190fecc9f93e82359551f27ab53db7d7:action", "state_id": "343c4b8b1f9ca37512f02c3424aede486bcf09b03e6fa31f00fabb1c8f6fdd19", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.92578125, -2.080078125, -4.4375, -0.4638671875], "student_probs": [0.9605657458305359, 0.006434428505599499, 0.0006091085379011929, 0.032390788197517395], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bf77600a151e41e3a4fb411cf70276f006c3eebb55070031c206870275e30bc5:action", "state_id": "d37f07bc9d3120c1cacf0fb4c1295715a3e03a65e87bd72f079b231d471a8988", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.85546875, -1.71875, -3.7734375, 0.078125], "student_probs": [0.9312379360198975, 0.009605118073523045, 0.001230731257237494, 0.057926274836063385], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "900fa0da24a357208c77eefc18a6a4f74f4e72d64c52c06c3d609c7962c3944c:action", "state_id": "f74dd4a0c5007f5d061bd46cab754509534802f034921faab798005ce528f5b8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.32421875, -2.23828125, -3.73828125, -0.658935546875], "student_probs": [0.8530007600784302, 0.024197770282626152, 0.005399251822382212, 0.11740225553512573], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9588ad4816bf48bcb0ef67b544d5760ee592e22412b064456e92b05dceb3e613:action", "state_id": "c90d39d1f43096461ab9cc46997cf4d3e6323f41d85a26d320e3e48559b7bd0a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.23046875, -1.44921875, -3.5546875, 0.125], "student_probs": [0.7102307677268982, 0.04871088266372681, 0.005932428874075413, 0.23512592911720276], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a1da441a46b6e3b2e92b692356a0488e2d7be4815f7013285b1869447ef80129:action", "state_id": "cd27cfc6105a05646d89c091fcad2b3729a99170d67533c35bc7588e83195a8d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.44140625, -1.546875, -3.3984375, 0.23046875], "student_probs": [0.5080649256706238, 0.06956961750984192, 0.010921851731836796, 0.4114435315132141], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3b43e30f38da210c773778901da3ecbb6a0222098684597f730174a5ffa80fd4:action", "state_id": "8122a4daf19a3b7f662618a84e0c0d23dea5896690358796ff277684cb4e0962", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.0546875, -0.83203125, -3.05859375, -0.05859375], "student_probs": [0.39915066957473755, 0.18345972895622253, 0.019795065745711327, 0.39759454131126404], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9b5099bb84850bc32609c5072bc3b07a75a6f40e65d1de0af04d46113bd28fce:action", "state_id": "28e182ad1c20c487f494666875206c327f42876d2aee22f633fe4d48bc054c50", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5048828125, -6.1015625, 3.1796875, -0.93359375], "student_probs": [0.0241062231361866, 8.943799912231043e-05, 0.9601027965545654, 0.015701545402407646], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5ddca5d274b6f54459adc5f8ce952893ce89533e9868b87e3a59d30bb430833b:action", "state_id": "36ac4b9302bef9d7dc7dc1a8d3bfb2016661e4c9245b926dc467cd32b3ecd57d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.19921875, -2.4609375, 1.76953125, 0.25390625], "student_probs": [0.10163519531488419, 0.010587469674646854, 0.7278826832771301, 0.15989460051059723], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2cb5f2d5295dd1f14796d38da18efaddb7bc1241325bbb5016be9f8205528dff:action", "state_id": "a148c0d07f07089a4724741796d33b78e3fc5915485112ea4914586c66db7e24", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3779296875, -2.58203125, 1.5, -0.462890625], "student_probs": [0.11670178920030594, 0.012877997942268848, 0.7632240056991577, 0.10719621181488037], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bc1fc7e13623002594c8575ebb43a648e4d118a4ef69b6727c4bf019d52c7e3a:action", "state_id": "326a20727fb9a6219ab85cd9073d09df580fc4d54679ac533649b6062514e0b1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.25, -0.90234375, -3.3359375, 0.62109375], "student_probs": [0.35804757475852966, 0.11310562491416931, 0.009921740740537643, 0.518925130367279], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3c6cd21898e9dd30b8071d9daf808dccd33f3883f5335c11488a7190d04de5f7:action", "state_id": "6cfdb892939f1810cedd2578a74d4abd690518ac1a3e1f1f4f81a37fce41d581", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.36328125, -0.818359375, -3.3671875, 0.33203125], "student_probs": [0.43478450179100037, 0.13338102400302887, 0.010426823981106281, 0.4214075803756714], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a3b48ae21e60366e53373ad5bee1852b36da0e1f9f95fa55f2e66d337b774571:action", "state_id": "09a7e97032a19649ed89fd047708eec5c634e9bd925a9fe92eedc69332d46161", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.19921875, -0.499267578125, -3.25, 0.80859375], "student_probs": [0.29687780141830444, 0.14764846861362457, 0.009431940503418446, 0.5460418462753296], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5451fc3896d56dac5d292f897f111ba877b03bc47217492a88fb60db0795953d:action", "state_id": "43d022d01217480376ffb4c02e220755368fb8ad42c71d81e28de35cda9f371e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.18359375, -0.44140625, -3.25, 0.62890625], "student_probs": [0.31964096426963806, 0.1710914820432663, 0.010315056890249252, 0.4989525079727173], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "003194b79b982452291bd9de5d3306d161ef4217d9b1bf580a785e6483faea98:action", "state_id": "b019dea72493d44055431e2e595a5074c93d62dd299273566c6b35c1fe6e5aad", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4638671875, -6.046875, 3.1796875, -0.87109375], "student_probs": [0.025064704939723015, 9.427426266483963e-05, 0.9581606388092041, 0.01668039709329605], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6ff6b45574493c426658fe244ec4210e48359afba4354f783f834dab6c4f672b:action", "state_id": "e2cc94f945ee51e8fc61918cf2a562b2d3f95dacb62751dba9e5bffe3196f267", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.185546875, -2.47265625, 1.86328125, 0.23046875], "student_probs": [0.09637372195720673, 0.00978767778724432, 0.7477447986602783, 0.14609386026859283], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e4380ab07e8e117512dffc5d88da913552df99df206f5f214e23e7a8a8d8cdfa:action", "state_id": "5a1992c8272521df705dc7505155f64989cc4b6e4c61da6a90e429f6d691084d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.181640625, -2.77734375, 1.79296875, -0.05078125], "student_probs": [0.10617732256650925, 0.007920129224658012, 0.7648807764053345, 0.12102171033620834], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8c25112bdacf90821f48b6d92705ee937139b59de8b957d473ab6dd05a4948cb:action", "state_id": "0a3bbee1c8602fd85bd361f565ec51d7a315a91c76bcf4c68f016bd0f41bd294", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.15234375, -1.96484375, 1.65234375, -0.015625], "student_probs": [0.11922045052051544, 0.01946220174431801, 0.7246304154396057, 0.13668692111968994], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c1bb9990eaf203d5c58d3a299fbb462961a63f95a1fc0d45ce067c749dcf7be1:action", "state_id": "cccb839725b4de7a9f6f47cbc3456fc8e72db6371c30526b3839ad3f9d14b1a4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.890625, -1.5, 1.447265625, 0.33203125], "student_probs": [0.29339396953582764, 0.026866799220442772, 0.5119141340255737, 0.1678251177072525], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dcbdc5e9086ff48776517f4159a251d7c205e7487c11ead89332cd5c4b33f9ba:action", "state_id": "be0354392b98a1c6c529352f9c7ec0c1891a1856164ca7387f25c5bff7d05046", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.40625, -1.49609375, -3.35546875, 0.46875], "student_probs": [0.4470359683036804, 0.06670602411031723, 0.010390794835984707, 0.47586730122566223], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "30c61429415b80bf3f83a5244815e0e560e4c8e0c795207ef947c98f7a57bc8b:action", "state_id": "35067e79aaf0e8b3ddf689a772f312805ea8cc5f53dbc3bb064e86436cd71523", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.931640625, -5.7578125, -7.296875, -0.21875], "student_probs": [0.32791003584861755, 0.0026289052329957485, 0.0005641162279061973, 0.6688969135284424], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3b8eefa028f236ab02659f8d228389f4d3f9aed91707507c77531deb41f671a2:action", "state_id": "80f809e17a81e4d241b102147897153fd2590f251f773ee7647dd3c20f62dd66", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.185546875, -2.93359375, -5.203125, 0.39453125], "student_probs": [0.35003572702407837, 0.02242078259587288, 0.0023174257948994637, 0.6252260804176331], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d2c7fc426db69cc132440eb6850e7448c4b488bf4890f9905379202bfbf9eb06:action", "state_id": "0d18cc39a0042a61c968e90f778534d7007dd70aaa952c1baa4651625ab8a95a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.369140625, -7.25390625, 3.78515625, -3.20703125], "student_probs": [0.0007801596657373011, 1.6034346117521636e-05, 0.9982864260673523, 0.0009174591396003962], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8713d4162e86715de92923477b583a83195d1433f0e8420126cefb7f09897880:action", "state_id": "41ca9eeba20a09422b3088a4c9dd465ea486842ff808c53e73a0c368849180de", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.70703125, -4.0, 3.22265625, -1.236328125], "student_probs": [0.00709026912227273, 0.0007158780354075134, 0.9808413982391357, 0.01135236769914627], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d9eeee8178f524dd3b0a17feac194bbeb565cc9fe882e94253767445a6b6b34e:action", "state_id": "62fc85b4ab6cbd77e101c283556a085c884bafa45573d86021377b1c9ba1976e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.630859375, -5.375, 3.3984375, -2.390625], "student_probs": [0.0023937264923006296, 0.00015392508066724986, 0.9944086670875549, 0.0030437360983341932], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b0f0f6876510333fe4d8c5713db47f203a1e4ccc3a6ca3c920a7a33fc047ee5a:action", "state_id": "039bc85d50f82d5de9e78cb8e1fd73a9522faf9216bac5453aa65b0deec06d2f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.64501953125, -2.625, 3.86328125, -0.248046875], "student_probs": [0.010707459412515163, 0.0014783996157348156, 0.9718887805938721, 0.015925366431474686], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ecbcbc5d2db8a50a1c8f43cc603cf38df0356ae7aa4e4e39d1f7741cd3fcc6de:action", "state_id": "1bf73e55c4ecbb37f8b9860a9c2d704c33490c7bb6208b6888ab1cd3e909a401", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6455078125, -2.58203125, 3.7109375, -0.265625], "student_probs": [0.012409140355885029, 0.001789452857337892, 0.9676579236984253, 0.01814356818795204], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6c753bcda1420a36c02632022d4898d934b95e403e1c88101481e0bde93f8458:action", "state_id": "73029ccdbcf18fa081440847bc864f4f8a86143f437010851c7ad9bf829552dd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.412109375, -0.267578125, 2.59375, 2.359375], "student_probs": [0.14235283434391022, 0.0265391543507576, 0.46403005719184875, 0.3670779764652252], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2a26375c76f5774f70ee3c72a1a5448bfe8403344c74eb29acf1b5f2ef52771e:action", "state_id": "021acb4d7b79c4b44ea2002f373c2a560ab1e3f6c8db4ca360237eab0271578d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.2265625, -1.36328125, -3.19921875, -0.146484375], "student_probs": [0.5194497108459473, 0.1059456542134285, 0.016894511878490448, 0.35771018266677856], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "80a75154a422fd6b32337f65f929976a6a60e9d3abdbf66b8723b89d570fa2f6:action", "state_id": "fe34977aa5d47e71d22745a7effcedc9cdb63539ac96c5a084c2536af3ee5ec5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.515625, -7.640625, 3.87109375, -3.474609375], "student_probs": [0.0016798427095636725, 9.988709280150943e-06, 0.9976663589477539, 0.0006438533891923726], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e4e152dd7f2e3d3b45243e825c83defa744cf5a490f18ed9d051c9fbaea207b1:action", "state_id": "954503c95b4f4b42bdeab67b9efd8def1e394d0a1432dee825ef26bca6dc4a6f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.671875, -4.51953125, 3.24609375, -1.396484375], "student_probs": [0.007189091760665178, 0.0004168239247519523, 0.9829257726669312, 0.009468358010053635], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bdf6ffd6a6768fd54d5e252318a5d49d5a1850ae307393f9bc587e19ca123001:action", "state_id": "eb83ec286c094ade0c8ee1654fe683d928361fa852c19db989cf451102ee5ade", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.72265625, -5.609375, 3.30078125, -2.451171875], "student_probs": [0.0024075298570096493, 0.00013424114149529487, 0.9942997694015503, 0.003158463165163994], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f4e0d8e53ea278b22ef656123127a2327d6aaa52c4326defa6b82d2b683049c1:action", "state_id": "785ca33061a1be0185944ee726c19b7b87473ac1fd4b73e39a584bfb62a2de01", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.77734375, -2.6328125, 3.70703125, -0.24609375], "student_probs": [0.010931473225355148, 0.0017094595823436975, 0.9687640070915222, 0.01859506219625473], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "631e1aafe9678572c3c3dddcf8054ef71484201549f5f2655bb118517dd5a5ae:action", "state_id": "c13a1d87d1a7e9610495b38b4b5ea103624cadd6de52202d8a4f8e8b8364b61d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.63671875, -2.63671875, 3.73828125, -0.26171875], "student_probs": [0.012190637178719044, 0.0016498233890160918, 0.9684222340583801, 0.01773727312684059], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "331986e1369b8d4592ee9d225b20199be5cda44dae741d27ae18ad88fffaf545:action", "state_id": "5258f1f6a1ee5cccf6b2ecc74876af32e05dfefe892783e64203ebaaa3703f9f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.16015625, -0.439453125, -3.16015625, 1.63671875], "student_probs": [0.16769973933696747, 0.092071533203125, 0.006060926243662834, 0.7341678142547607], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2b5ef0b476fb2a7c496bd107bd4fcc466cd332ac553c11373857c7208a54ef5b:action", "state_id": "3ba956e0fc0aae8a01c008014d4c72ab3a252c8eb5661f78f0e9d29d538ed00b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.28125, -1.27734375, -3.19140625, -0.0703125], "student_probs": [0.5141245126724243, 0.10818813741207123, 0.015955589711666107, 0.36173176765441895], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "020ecc4b80431dd84d637c8b2300a4f200c3a17641a6c5228eae440f36ff518c:action", "state_id": "c514855b4bebca713f3b679d79d9cb3e9bd0256bef74969bdd17497a9c67e00e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.564453125, -6.1015625, 3.1328125, -2.626953125], "student_probs": [0.0033331129234284163, 9.698463691165671e-05, 0.9934386610984802, 0.0031311700586229563], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0774b3f2226a41631d6f5b7adc5b79378158c8872dcd7cd12c1a2d212ee1b633:action", "state_id": "b40aa0126469882f5c66bc6b80828ed8e97e6095fe10200557284d9f3720c1c4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.81640625, -2.796875, 2.55859375, -0.798828125], "student_probs": [0.011964375153183937, 0.00448825815692544, 0.9504480361938477, 0.033099282532930374], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6fc225db8e20b08c33727ad835fa49c740208954b05bbd1fe8f54e61d5bffb5e:action", "state_id": "b5eb857354ea31864518d4e75cd5efcc7b91c0986ea049d0208559a3af87b934", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.99609375, -3.1796875, 3.3046875, -1.287109375], "student_probs": [0.004906014073640108, 0.0015021058497950435, 0.9836233258247375, 0.009968659840524197], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "de8f9a1131c08b0762433674abc49ecee0af1276223c68106f42d58fec57543d:action", "state_id": "4401808462ced42994bc95f01e437f9fe3ed40dce68e873b6eb5e2b4af95035d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.296875, -2.48046875, 3.77734375, -0.8671875], "student_probs": [0.006146651227027178, 0.0018819597316905856, 0.9825253486633301, 0.009446033276617527], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5008b4d5bb9ccbc5ce38587c87549f68ae599e7432408e7e9044010661cc8c12:action", "state_id": "8f7a6c4095cb1bc71ff04ef6002fc93ee15e5fb396155f3113c5580386112b73", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.796875, -2.52734375, 3.943359375, -0.3955078125], "student_probs": [0.008537369780242443, 0.0015128332888707519, 0.9771960973739624, 0.0127536840736866], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "86ff92bf46d55a9c049ef941f7256b97ae201a7d67a538488986d9c97dace7b4:action", "state_id": "25a4bc00d36886daa9189c151c5946519a85ccc9df73aecda0c1e45f66d02e44", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.07421875, -2.796875, 3.89453125, -0.57757568359375], "student_probs": [0.006818084511905909, 0.0012176495511084795, 0.9807608127593994, 0.011203449219465256], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bb442dd9f8c4fac95c1dd0df60f46b5be24a16477009aa8a2b8041e20836e95d:action", "state_id": "5b53427cf0ee127c03676925966ea0b7082a94ef4c15b6f9593feb758b1f1731", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.794921875, -2.40234375, 3.630859375, -0.3271484375], "student_probs": [0.011577434837818146, 0.0023201596923172474, 0.9676197171211243, 0.01848262920975685], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "53859808b2c6512ca8484ee908b4fcced0c2b656b0e8c5e377526d1a1f7ed980:action", "state_id": "dfea81457cd1ba6bbd350a09e639f03384f95bea33bd6a7433a0e25f08e2e3df", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.86328125, -0.09765625, 2.58984375, 2.029296875], "student_probs": [0.22782789170742035, 0.03206140547990799, 0.47113892436027527, 0.26897168159484863], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "73d75a3c4ead8368be63d1dc27d85869f38e1fd069d607daaeeb42baa741fa45:action", "state_id": "cded300eda31d3e89c6cab933befc4a3313655407e1b32c9e8d686274c0555d7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.2265625, -0.294921875, -3.03515625, 0.71484375], "student_probs": [0.30660977959632874, 0.18201543390750885, 0.011750045232474804, 0.499624639749527], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1384e87f23d616ee743dce61aaa7f1a2da20cc9c5b8e798e51a5020c2d816fc2:action", "state_id": "19fdce8af7c6aa8cca6ef26575144d8ac5d59bb328f527fc68c3dbe10785fd74", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4609375, -6.296875, -7.7890625, -0.4873046875], "student_probs": [0.2734279930591583, 0.00217081094160676, 0.0004881724016740918, 0.7239129543304443], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b6f4ad73b0d460ba4fadb8fa8bef7631c2c6fd62d6c62333d9d179e1ddc9f0f6:action", "state_id": "93085388e93e7108ea8c1e2eec6c080a79edd9b966738a7fce2bda8c3790795a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.65380859375, -3.25, -5.7109375, 0.58203125], "student_probs": [0.22113187611103058, 0.01648692786693573, 0.0014072400517761707, 0.7609739303588867], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cb49b29021edb34ae3b1b6602f39094378c7ec99ea7446e80e7c07661be9b6c7:action", "state_id": "04570377e588299d0183be310f8bf0144415d206934d8bd887f63fcd4bda266a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.7890625, -7.7578125, 3.8046875, -3.57421875], "student_probs": [0.0013661609264090657, 9.497322025708854e-06, 0.9980012774467468, 0.0006230355356819928], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fc60a51ac7fbb776c759c088429a712f0779c619eec106caa4109bb1b4c7e5b4:action", "state_id": "79b12947fa03fb798191c2195ac78cfcc1133787f7dbb4427b4ca7f77899df4c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.849609375, -6.796875, 3.2265625, -2.623046875], "student_probs": [0.0022850199602544308, 4.411784539115615e-05, 0.9948047995567322, 0.0028660567477345467], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5863813f3e22864cfcb6d11755b8a8c3ae6135fbbee4dc179b94c460c0f4c335:action", "state_id": "d8f7a53524ac403e853b3e914c94526293ad421c7460a6fc206be7bc37b72506", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.478515625, -4.05078125, 3.08203125, -1.08984375], "student_probs": [0.010184632614254951, 0.0007777223363518715, 0.9740151166915894, 0.015022541396319866], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8af160cf72dc39aa6c6acfdaf9423e5dbb7208d620672fe9d0ced4ca27aaffd4:action", "state_id": "35bd07607ba1dfcebf62fa12fb0c21b3ecac475e976620d9923165d7af3237a9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.26953125, -3.07421875, 1.85546875, -0.88671875], "student_probs": [0.03938430920243263, 0.0064797368831932545, 0.8963826894760132, 0.057753268629312515], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c327a717e789d73a6f3affd7bae773149ad3a9e4b118ae634e970e45bca5c8b3:action", "state_id": "6bb7a844568956089b5cdc167c5d896ba82fa0bf206bf77a9c48c185811ddb82", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4365234375, -1.56640625, 1.890625, 0.13671875], "student_probs": [0.07493019849061966, 0.02420778200030327, 0.7679352164268494, 0.13292686641216278], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5854c0a09cb9c7993f320e1be8ab218c11d8a0f8e385242e56e34584db2280ea:action", "state_id": "d1b34e2d96e58cd9592ee2b26b4857cefc8bda83e72e85ece10306cc069ea572", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3828125, -0.3671875, 1.509765625, 1.560546875], "student_probs": [0.12811289727687836, 0.06051625311374664, 0.3953869938850403, 0.41598379611968994], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fe02307463a764792550d5b0787aa33f6950f288a102a84f295ee1fde1593e18:action", "state_id": "09804a43da4c1f2a4d35294989961e7e610c83364a9a826f8b0e30e4040dccf2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.43359375, -0.091796875, -2.57421875, 1.33984375], "student_probs": [0.24296921491622925, 0.1436736136674881, 0.012002588249742985, 0.6013545989990234], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4214c20927c1d951aec75f28195b0e344607930fd0799860f1d29823350241a4:action", "state_id": "2f81a150621b34b6a4d87573c1e69f095e8468a33cbcda78fbacf17eb8fb8496", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.00390625, -0.34375, -2.54296875, 0.4765625], "student_probs": [0.29345703125, 0.2089066207408905, 0.02316560409963131, 0.47447070479393005], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b73061aadcbefa7bb341db5f6f46cd0f57ebdad8b1ad284859131b208705980f:action", "state_id": "e70603202fb711eabc52b9e24003449f9a243098bc17cffd9f43df25e24aba18", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.974609375, -7.58203125, 3.7734375, -3.62890625], "student_probs": [0.0004311306111048907, 1.1692985935951583e-05, 0.9989479184150696, 0.0006091802497394383], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "84c291fe5c1fd85ac4b12709a9a9014bb0115cfa582c4e2087f2d79c5736b039:action", "state_id": "09b3d1b29dd4008f058f481dfb37df7edc9ac7ea22f32cbbf79e6d19681228d5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.998046875, -4.1875, 3.30078125, -1.447265625], "student_probs": [0.004927352070808411, 0.0005517548997886479, 0.9859738349914551, 0.008547022938728333], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5d872b6c2d687fba67e647f52893a691db5de58cc22132cdbabdaaebaa885bd6:action", "state_id": "d6b824c817f8a8c88e0dfb574aa0936cb4d10d387e517585094bc1f7f24352f9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.87890625, -3.75390625, 3.30859375, -1.62890625], "student_probs": [0.005510922987014055, 0.0008451273897662759, 0.9865678548812866, 0.007076165173202753], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "284e8d602da80c4fa861246702f63926e3922cb1716e2587641f75901702d134:action", "state_id": "5388cae9b62add7392595a8e95d8644aaee2c78b6965ba6ea445d04c7112c594", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.19140625, -2.8984375, 3.14453125, -0.7724609375], "student_probs": [0.012642495334148407, 0.002293393248692155, 0.9658429622650146, 0.01922110840678215], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dd18526bd655bd29cdc4a36bf814ebca6856ee8d0a03ab58795ba7f5d6618e14:action", "state_id": "c7ff7bd3361629ea01f50b35c2cf1d5c9774e6054f7a7fe3c06396ac1b62c125", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4599609375, -2.078125, 2.390625, 0.2109375], "student_probs": [0.04889456927776337, 0.009693953208625317, 0.8457740545272827, 0.09563747048377991], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "352c28d9409efac128709e67e919320020eb67e7de89833df7cff2ab2f464b60:action", "state_id": "ddf8cf8452f0fb266cd50cf4d73560fbf642b049bce264d60bb756658756057a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.044921875, -2.42578125, 2.41796875, -0.216796875], "student_probs": [0.028209170326590538, 0.0070907254703342915, 0.9001286625862122, 0.06457142531871796], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fffd5abfe7871a6e10a43f5be50e930df9a76ea3e7ab04b87af6bb283d8fc0bd:action", "state_id": "513dee3544c077286b11d57ea40f165d332db77ac08d7d7f963ebde11a887054", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.078125, -0.939453125, 2.45703125, 1.0078125], "student_probs": [0.058812398463487625, 0.02485414408147335, 0.7421184778213501, 0.17421500384807587], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f147cc9f4bdb19233e90d38fb093586deb837cc2f65fe2a0df6c3d85434cd847:action", "state_id": "39a00ced08dc668e3fdc9a4f3f010a2e5833a28bd094f6fa3b6abc21a6505c35", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.36328125, 1.091796875, 2.7841796875, 2.218017578125], "student_probs": [0.12115571647882462, 0.09235060214996338, 0.5016863346099854, 0.28480735421180725], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "797b0cd14da6772841e8f524f172de77b6c04e462b5afc313977aafcbaa507a9:action", "state_id": "dd1194665d4f798407897b87e9e8c1835a618820520a95d251cc8c7156b30345", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.43359375, 0.34765625, -2.2109375, 0.8515625], "student_probs": [0.28509819507598877, 0.26162081956863403, 0.02025299146771431, 0.4330280125141144], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "06eca35fbfffa53b56ee677cfb87027e219540e3c2b4fc1b0f69f5e917877c16:action", "state_id": "c045f484559deffe91cdf38dc35411987be8866a8bb40e228cb1843ee4454caa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-4.06640625, -7.62890625, 3.8359375, -3.66015625], "student_probs": [0.000369529880117625, 1.0482755897101015e-05, 0.9990652203559875, 0.000554730067960918], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5298bac1c43e9c61d5212733a3c7a5176b0984a2defad77b9bc7f749579328f6:action", "state_id": "df393c104241fcd34c588ecf8e6d6e8f7c726122eff7e9301ac878c273d9dbdd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.009765625, -4.32421875, 3.36328125, -1.44921875], "student_probs": [0.0045794048346579075, 0.00045253775897435844, 0.9869465827941895, 0.00802142359316349], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "639f69756f004f70d8506d5464ddb9c841f36f21f415e159712b7079493b43b1:action", "state_id": "d1e88b1729161b5edbfe9b10314c5102d134f91fa6ac1c0682db3ff4ffc61739", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.51953125, -5.59375, 3.33203125, -2.865234375], "student_probs": [0.0010544003453105688, 0.00013249021139927208, 0.9967846870422363, 0.0020284445490688086], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "12fe72738e248824e13a4efa0a44b5e43cc55ad819c1269a7025113b84c4c0d6:action", "state_id": "d62cdc60c90b92eca5a6237863dbc4f15eb9770663d878f95745385970ec9cdb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.056640625, -2.98828125, 3.78125, -0.7578125], "student_probs": [0.007770246360450983, 0.0011259884340688586, 0.9806273579597473, 0.010476451367139816], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "df10c51d7ff98ddd3f475e1b8d047cc6f855db2afe33e06047222edeb58e231c:action", "state_id": "e6e2f25d569be03597aca4df7aaee346d4ea1c0a605225c073e69ffdcc370f63", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.751953125, -2.5078125, 3.9140625, -0.3046875], "student_probs": [0.009173449128866196, 0.0015847933245822787, 0.9748942255973816, 0.014347546733915806], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3e94ec61e9be8ea0a3309b725139cdf1b270ed1678d2455f46f9cdcdc9c36444:action", "state_id": "ea166b0a24865b9500ef5ad5a88cdefdb9ae1f55b7a97ba7e65e53d3dac0e65a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.596435546875, -2.4140625, 3.76171875, -0.3876953125], "student_probs": [0.012421224266290665, 0.0020173396915197372, 0.9702569842338562, 0.015304499305784702], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cef40435e13b300c25fdc1467712e94dcc052f6334651cc1a9bbef309f5a43ba:action", "state_id": "6f56f41f45af30c29085c824596dba60fb518f2c73faa017a99840bdfcd83073", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.46875, -0.216796875, 2.373046875, 2.287109375], "student_probs": [0.16885288059711456, 0.03129570931196213, 0.4170994460582733, 0.3827519714832306], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "86d0e72725fc2844e3e17539d81a4ded113ab025cef552e50078c918f36b1e4d:action", "state_id": "cb8aa09845d7e678585978e62f0638c172b44c95874d92b12b70cfab2cf737a6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.38671875, -0.7734375, -2.86328125, 0.1328125], "student_probs": [0.46993011236190796, 0.1472935825586319, 0.018221167847514153, 0.36455512046813965], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f741da7ec1792e4d5ab8e14f65d6984751af70c133c6504b3c9d4658bdd418d5:action", "state_id": "fbedaf26897a35c7b3e5f88698cee5222724a4a70f95562038e43d62688e625f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.8203125, -6.59375, -7.9375, -2.677734375], "student_probs": [0.9958200454711914, 8.122794679366052e-05, 2.1189576727920212e-05, 0.004077645484358072], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cf38a8ebd05a296e180303e6c9e72b9940463c45fb4326e81c36d83941f498ec:action", "state_id": "3e1acec01a35fad5456424a6e24d05ce1201c203e3c1bcfc247c9f95705ec059", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.28515625, -5.6328125, -7.6640625, -2.373046875], "student_probs": [0.9902015328407288, 0.0003605732345022261, 4.729691499960609e-05, 0.009390563704073429], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d72d08b3062a0288da6c7d8a3eff40f2b25675116167f4f22cf3a6913532cddb:action", "state_id": "578c4b567820994029de250658a41a0c77c01323236ffc400d22b19c661f7813", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.3984375, -4.09375, -6.375, -1.564453125], "student_probs": [0.9797408580780029, 0.0014845335390418768, 0.00015165464719757438, 0.018623001873493195], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "84922c91c7ee8e933e5cfc48b4085f26315f7ee0e426853b2be1b3ac1ff29483:action", "state_id": "cc4c6cffd31c57670342d596b47b89f08ec54ef709a348dafe46eedd300a7cd1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.6171875, -2.625, -4.796875, -0.322265625], "student_probs": [0.944477915763855, 0.004995036870241165, 0.0005692530539818108, 0.04995782673358917], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "770b49f75caf00b0bf6e3ffe3a2a55f45f68ed3612f274e837d8bd566364d8cc:action", "state_id": "1091b87998b65d54506f1ef60f753f7b202f8b80628bc299e12cae0b13ed46c4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.2109375, -2.33984375, -3.85546875, 0.44921875], "student_probs": [0.8441470265388489, 0.008913307450711727, 0.0019579939544200897, 0.1449817568063736], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a54d454631523731eed7701d969d85cc33b083c8be9fcf428d2a064d2a9df386:action", "state_id": "cea7fc8679bb6b944e0dc86b2f3048a5fc5924870bb359f6e11942e0b9e8c7a5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.578125, -1.59375, -3.4296875, 0.8828125], "student_probs": [0.4018746614456177, 0.0457991361618042, 0.007303310092538595, 0.5450228452682495], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "08b7c3512efa4f2402d000d9cbeb8ab03e8e3e1f34d5b19c0a18031637306c3b:action", "state_id": "ad01fe34e2497d09eb0a5d875437896fa53497f37fbc9c075bd02c653a95acbc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.24609375, -2.23046875, -3.328125, -0.4921875], "student_probs": [0.6289294362068176, 0.052849940955638885, 0.01763349585235119, 0.30058717727661133], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0772eb78fc545014ccd12fa0937f96b13f2e36481f06cfb2aaae65dcde40f572:action", "state_id": "479b27ef1f2a0b0961f5723dd306e68ccb69c30a732c62fdeda8b390baa4b182", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.8359375, -7.59765625, 3.80859375, -3.525390625], "student_probs": [0.001298566348850727, 1.1103910765086766e-05, 0.9980387091636658, 0.0006516860448755324], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "74e149350e45d8a0c3ad56d170c0851c9a332e09d056e5a17acc4692c77056a6:action", "state_id": "9288d02d703d39722e06b47348b79760cdef037921dae60f1bc55a02cc22e1a1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.71875, -4.32421875, 3.078125, -1.408203125], "student_probs": [0.008092622272670269, 0.0005977898836135864, 0.9802697896957397, 0.011039720848202705], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "12a27581a57cf313c61779b0113819921c675ea3c8555b37206f821005d195cb:action", "state_id": "35512aa9b306522dabb2f6389d02d4af430a2fb3d28997b1f622aa40abbd7b7e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.583984375, -4.12890625, 2.953125, -1.51953125], "student_probs": [0.010464036837220192, 0.0008212091051973403, 0.9775541424751282, 0.011160685680806637], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3f2021782c5762cc11e0eedca1a09d95da2f8bf5b76dd2bee706bc3f4f77e0f8:action", "state_id": "58b53bae8b7a0ac1a62f461140e78aa111dafb190b02245884aad6029db040e7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.1953125, -1.37890625, 1.744140625, 0.2890625], "student_probs": [0.14262473583221436, 0.029547473415732384, 0.6711851954460144, 0.1566426306962967], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "231ba1b4e13c9489c5523556eac6d0e04ac1b88b620beed14a363508850784a7:action", "state_id": "db428090da3b17b276a25d1930548ee5db44a101f84c6881df1aea89edc4a7a5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.02734375, -1.4765625, 1.69921875, 0.39453125], "student_probs": [0.1251867413520813, 0.02782403863966465, 0.6662610769271851, 0.18072816729545593], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7d4b56f3d5af5e5fb04dfbe769ba869ce959042edc3a628ebfe02a89c3d53bc5:action", "state_id": "d29d7623b1659f33e151f6ae95b31bba2a63d3c1d5d222d08ffedeff38289221", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.091796875, -0.984375, -2.90234375, 0.5], "student_probs": [0.3051568567752838, 0.1249917522072792, 0.01836192049086094, 0.5514894723892212], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c46dc8c94e59263810bbe64ea30702a3891139f7ea262fb69aac22887a83cbe7:action", "state_id": "4b217c420309a55c2f702353809b0be8501fa708cefbe63c014abb5e9b2d0dee", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.37109375, -1.6796875, -2.9140625, -0.197265625], "student_probs": [0.3939049541950226, 0.10643302649259567, 0.030973775312304497, 0.4686882197856903], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "11977eb5564d251533ffab9c351c5aed99d155e0ed1ab7c7ae932d376279f712:action", "state_id": "d00d4908e371d85cb1269c39fa227f47ec3133f2db1e550adba8ecc2d730aef3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.03515625, -0.365234375, -2.43359375, 0.62109375], "student_probs": [0.2675744295120239, 0.1923505812883377, 0.024311764165759087, 0.5157632231712341], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b0afdcf653b2b2ef0aca826062b050725998623993f15fb4426b39d50c7094d4:action", "state_id": "913add7039aae993a9a7cbe271e14bafde42153d7b1693966193d1bd33138dcb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.5234375, -6.5, -7.6953125, -1.638671875], "student_probs": [0.9845119714736938, 0.00011868392175529152, 3.591486893128604e-05, 0.015333449468016624], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "28b0e21e2f9abee11a4c5afbb46a88259f7910d575b532b83ef3c70ab084d6cd:action", "state_id": "900af2d4aada801c637e020f167d3ad8305b718f8b0cdc68cc59d1d0843dd461", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.2734375, -5.015625, -6.96875, -1.599609375], "student_probs": [0.9788808822631836, 0.0006685443804599345, 9.481974848313257e-05, 0.020355742424726486], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "12adf5388ca852d01e97cd5463034a56fe14dfcc08ee90e3703ec6cfe9936504:action", "state_id": "4c15935c2840ed4b3708d2ee7a1c30d387395d9e70a71af49c95af1ca2bf9b84", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.8984375, -3.421875, -5.671875, -0.7333984375], "student_probs": [0.9282007813453674, 0.004540039226412773, 0.0004785165947396308, 0.06678056716918945], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9a7acbf26dd7072a685df8210f6e18ea3e8e05c361dc43e832e633d4006cbed4:action", "state_id": "0031eeab567f899a98560bec555e9d3f7bbc552282139a4f8b104b980e870414", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.01171875, -2.2890625, -4.7890625, 0.82421875], "student_probs": [0.29744741320610046, 0.029798446223139763, 0.002446005353704095, 0.6703081130981445], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1bcd9dbbcf3774f22baac45a106bd5dc9abe3d4f02815f54a5d59397142829cf:action", "state_id": "b3a7c816dd4ce30df167487788dd193cf726c82909684e4f21dba11b43be831e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3984375, -1.8828125, -3.8203125, 0.11328125], "student_probs": [0.5351113080978394, 0.054665062576532364, 0.007875248789787292, 0.40234845876693726], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a888c064e4e9a8a2d857646b38d73b76af38ecd9c94a3c7a65feee6b974cde69:action", "state_id": "f116b31d2092f695e6af53e902cfff0651119dfb6f86b1375cf894cc87462bcd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.4453125, -1.5625, -3.58203125, 0.35546875], "student_probs": [0.4839855134487152, 0.0649905875325203, 0.008625399321317673, 0.4423985779285431], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "edc68dd2c0e9d289865f41934c21a9ae9ed842316644b5f53dd5dae53365f7e1:action", "state_id": "ed21114e18173cbb591c5cb64179e54f181c88719728696b5c72c14b264ed033", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.1640625, -1.86328125, -3.5859375, 0.0234375], "student_probs": [0.49406686425209045, 0.06506111472845078, 0.01161933969706297, 0.4292527437210083], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ff980899a90e31a1db2cdcd93f68b9417d1bc58d2834ddce988580f68f8c1cee:action", "state_id": "e5cb3e5e8e5e81a939d54de9ccb31789d60c69d6a1a523f8c06f138071080135", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.43359375, -6.046875, 3.1796875, -0.8779296875], "student_probs": [0.025818148627877235, 9.421239519724622e-05, 0.9575318098068237, 0.01655588671565056], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f519fa5b086d110ff0db2d7a5b4d317d1abc73964654575e49065e564c39e9f3:action", "state_id": "e99bbaefb20ff130a1731843ad11ab006721f3c11d578b49eca2d48a1259d485", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.197265625, -2.5078125, 1.91015625, 0.21484375], "student_probs": [0.09228318184614182, 0.009155136533081532, 0.7592141032218933, 0.13934756815433502], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "172570e12838d3c281544ad00a69725c0461d1014387f3dd07ab0e0abbb08eeb:action", "state_id": "0efdb2c53796128af14120f4d329e682d14275ab16d9041dac9f97329da66f2a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.443359375, -2.62890625, 1.640625, -0.48876953125], "student_probs": [0.0989663302898407, 0.011125423014163971, 0.7953355312347412, 0.0945727676153183], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "578af901d57bc2a593791c661d0f4c32cc361ada2e40c76147b721daa1056194:action", "state_id": "abe040e7595ac6cb1a7cb00fd9a94953cefc4e0a4623d9726d4fb55992f0b0b3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.923828125, -1.142578125, 2.330078125, 0.91015625], "student_probs": [0.16145475208759308, 0.020446643233299255, 0.6588361859321594, 0.15926237404346466], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cb693ae0e3c903d36d3dd67cc0b5861ce07c9c2ae64a3f08a46b5984e2d409d4:action", "state_id": "87d5217051db75bcc8957737e31736b1375c9f5ce9fe7720a2a1ac0e0074769b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.25, -1.140625, -2.9375, 0.171875], "student_probs": [0.4514583945274353, 0.1123768761754036, 0.01863391324877739, 0.41753074526786804], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "db9e3ec63daab1b5254672d3aeed55b949f000c4a22e14809e9aca02e68263d1:action", "state_id": "88972aa21de650020710b12337b3e715f99ce38ac7048ccfabd23aff785b3392", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.7109375, -6.1640625, 3.1484375, -2.7578125], "student_probs": [0.002836952917277813, 8.977987454272807e-05, 0.9943662881851196, 0.0027070397045463324], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "235ce86972cad3b7090cce4047e3bcdf917945bcb156c77c76d18dcd61a95a39:action", "state_id": "81a95b29b4c9e757b57876abddca8d9c4938c0dd2cbbe5f8161e6c2325897191", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8125, -2.6171875, 2.140625, -0.9140625], "student_probs": [0.01785685308277607, 0.007986078038811684, 0.9303048253059387, 0.043852198868989944], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3a78b89899c549d543c54b58111fc6acea35f4358c5a028d76c827c87b62f024:action", "state_id": "d4b5ae14e3bb98be0de2a7c11d82ccd50356919484928a556dd91cc88dc5bc67", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.25390625, -4.65234375, 3.33984375, -2.205078125], "student_probs": [0.0013612546026706696, 0.0003362061979714781, 0.9944171905517578, 0.003885434940457344], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f42afee3da8fcb618b84ea22350ed2a053f52aeda00748931972cbbba76913d9:action", "state_id": "7b2c324aafc27355155453bb4711c358bb3fa5ca73a3bb2bd804c05f36939042", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.265625, -2.48046875, 3.78125, -0.8828125], "student_probs": [0.006317167077213526, 0.0018746595596894622, 0.9825447201728821, 0.009263512678444386], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7473a0c9d27b6fd73eacdce72240fb69d915262ff041573f12211355793239b5:action", "state_id": "fe5ac2c013bbd97d7df4982f8fe5e222d09406c9d2e56ed49e5ed4048ee6fccb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8203125, -2.4140625, 3.953125, -0.384765625], "student_probs": [0.008259394206106663, 0.0016779976431280375, 0.9772951602935791, 0.01276743970811367], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "eee165553d28f2f7a8bac1d933675c19665a8fde1cd15b9ac4abf5cecdb8986b:action", "state_id": "615734be029853fa5e7f57ffbb83bef5a8d1d19536e1549188341464e2cbc1e2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9765625, -2.58984375, 3.90625, -0.53076171875], "student_probs": [0.007420468609780073, 0.0014784007798880339, 0.9795122146606445, 0.01158884447067976], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "41ea2a43763c7d524e94579406ea8ee60d34411d2cfbcc516c304d45787ceeca:action", "state_id": "0cab6890d050220854382237afc86a88390ce0e6ebbe0242944557e7264229e5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7958984375, -2.48828125, 3.595703125, -0.39453125], "student_probs": [0.011983565986156464, 0.0022059392649680376, 0.9679086804389954, 0.017901837825775146], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9a65f42cc2563eb6230ab66c4383bb7a6d3508c62db0b935ac18507d8a4a2b01:action", "state_id": "ca0da50cdc245b6db85fc34b8dcfb56ddcaa27e936f1cddab16021f64b7058d5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.94921875, -0.0078125, 2.6083984375, 2.080078125], "student_probs": [0.23728786408901215, 0.03352336958050728, 0.45872625708580017, 0.2704624831676483], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ec5cab1a84d45af039624b82435fc31e8417aa39f5357e1d41ac116e724b0222:action", "state_id": "abfbf8a55c1a11dd8cb783e57da31d49df93c1ac4c4ea17610fede5fe7dd5316", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.36328125, -0.259765625, -2.7890625, 0.76953125], "student_probs": [0.32465028762817383, 0.17411251366138458, 0.013879387639462948, 0.4873577952384949], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c92fbec849fab6075cc8e10b26f40e9d20c3496b75c6f6aa48a96fc8fb623be6:action", "state_id": "f54ab505c45f5a5b084a985ba5299cf189e43b07e83756baa490b314011289fa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.8203125, -6.58984375, -7.9296875, -2.66015625], "student_probs": [0.9957475066184998, 8.153992530424148e-05, 2.1354211639845744e-05, 0.004149653948843479], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7c468ca8fe09ece49a52cfe3f5219a1126e3844649f093b8dccd28acaf9032d3:action", "state_id": "569a223901fc6cce34fdeac536c3604b9916ddf1e79708837ffe781de721f92e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.30078125, -5.8984375, -7.7109375, -2.33203125], "student_probs": [0.9900528788566589, 0.00027213405701331794, 4.442466888576746e-05, 0.009630603715777397], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "499cac50624990204bb3119b021d3896e146637da3d92468bb273abb5960393f:action", "state_id": "d62a5c2db6e3aa189adeec2c1b70ba6db3f276b6d8e604e1009b4c5c6ac756e2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.2265625, -3.6796875, -5.5390625, -1.1640625], "student_probs": [0.9644745588302612, 0.0026256630662828684, 0.00040899941814132035, 0.032490845769643784], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d3f288f250ec73f346e97465fad00e21afc6176af0aa21646d7365be1641e130:action", "state_id": "68c05a1a8b17ca13ee7ccdc893d07c997e82b422bce5d203d418cef98ffa8d1e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.30078125, -2.81640625, -4.6171875, -0.751953125], "student_probs": [0.948575496673584, 0.005684674251824617, 0.0009389365441165864, 0.04480084404349327], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c100d9f2d03a71d0204e92dc6dd9875e4b3a66ab07010a76f203f4919d8ac582:action", "state_id": "b535db41f9b04d260ec0e4f35001e8c45de2c826b3713714765effca01ee3e2e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.796875, -2.14453125, -3.79296875, -0.150390625], "student_probs": [0.9425054788589478, 0.006733772344887257, 0.0012952425749972463, 0.04946553334593773], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6a1b5ea98d91281844cf00a7385783013dbe9b4dccea019090c94efec4c06947:action", "state_id": "0272da34ebd9c091b0c01fad330ecdbf674a92adb003b33344cbb3d49d0ba890", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.62109375, -2.02734375, -3.7734375, 0.34375], "student_probs": [0.7637377381324768, 0.019881445914506912, 0.003468399401754141, 0.21291238069534302], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4f122109fb5df5e665fd5d605a0f312486582a39872a40a3b5d06699d5cde967:action", "state_id": "53345d424dc33444856ec01fa3ad6fa37866d6eda2f9b7a362300e883e927413", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.40625, -1.26171875, -3.47265625, 0.91015625], "student_probs": [0.3491038382053375, 0.06585139781236649, 0.007217171136289835, 0.5778276324272156], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b63601549e6a8e01cedf77db34c74357c12dcf0371bd9bc52691d3a259b2b1f1:action", "state_id": "41307f99370c92d4bd41a89a1839c450846c4e3ed29f6f879babf8a32d441f34", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3203125, -1.515625, -3.25390625, -0.166015625], "student_probs": [0.5548215508460999, 0.08847403526306152, 0.015555709600448608, 0.3411486744880676], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "08f5e5ca673a4427220ebbb5d435a8ac189ef6653cfda318bb313701e68c9428:action", "state_id": "84612cb2c0ca371373ba2005fa1c3ec2383eb671b6cba632c5878106d71b1270", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.564453125, -6.1015625, 3.1328125, -2.626953125], "student_probs": [0.0033331129234284163, 9.698463691165671e-05, 0.9934386610984802, 0.0031311700586229563], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "35405795abf80599743f987857a8f73e29217c2cc64c07e4fe9d9994efdfc367:action", "state_id": "1c3280bd035fcc4d9d9808dd9240cfd690fb6082dd62ba76a1e2d7b94b3b351f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.81640625, -2.796875, 2.55859375, -0.798828125], "student_probs": [0.011964375153183937, 0.00448825815692544, 0.9504480361938477, 0.033099282532930374], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f855ae31148858cf2f1980acfec1baa3e2525826c5093becea7bdf12fe734634:action", "state_id": "733f9b5a44bab6835f59819f463a1cf066ca4c830ba5f6afaa7ef59faace2f63", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.99609375, -3.1796875, 3.3046875, -1.287109375], "student_probs": [0.004906014073640108, 0.0015021058497950435, 0.9836233258247375, 0.009968659840524197], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ad7488129b06f1039050f2a0b9b9dc8451d275627e95dd0380c6cf57c1995583:action", "state_id": "66d95a6b63665238f120827a5480ddbe6e5b957da944f0114efa32dfc7dd1fe3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3046875, -2.51171875, 3.80859375, -0.89453125], "student_probs": [0.005916424561291933, 0.0017695071874186397, 0.9833976626396179, 0.008916366845369339], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fac41c4d5d988dbeb763becc18edd083ead722bce606dfd070da8cddb92e24e9:action", "state_id": "a27df0055bb762b3f64daf6230cc1509bafb9beafc01d5b6713ae8eba579f5a7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8515625, -2.59375, 3.90625, -0.4345703125], "student_probs": [0.00839043315500021, 0.0014694741694256663, 0.9774084687232971, 0.012731565162539482], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c03878b8e605f18ff49340ccf71244088707fa3841c0ca8432b0a8c34d4999ca:action", "state_id": "89b9d2a00ac60f1b99d6f3d7c128fcd9bc9442e749ba2594fbab350d0806ae1a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.07421875, -2.796875, 3.89453125, -0.57940673828125], "student_probs": [0.006818224210292101, 0.0012176744639873505, 0.9807808995246887, 0.011183183640241623], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c802fb7f92e5a93f7f19a27014b80546cceacc7656580b5724184cef779323cd:action", "state_id": "e43298634359d43b35b043d45b745e1062de1dd1b94c8039f190b1cf04a0a2bd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.80078125, -2.41015625, 3.630859375, -0.3330078125], "student_probs": [0.011512026190757751, 0.0023025500122457743, 0.9678071737289429, 0.018378209322690964], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "baf4b1b97408eb750c1a4a7477577677cad1c9a7530e89e0c52c24e402bebf4a:action", "state_id": "016f9a7f9db1ee2ee3e3b4f0116674b965d16439a7430d230ba25d748fb0d41d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.8671875, -0.078125, 2.5947265625, 2.05078125], "student_probs": [0.22652876377105713, 0.0323805995285511, 0.4689100682735443, 0.27218055725097656], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6f782fce3d6c03b7759a0a1aeecaa4093bbd428dfb881d6a9762de520a799596:action", "state_id": "7773223a596bddde217dec33309b5c9dad97ad77fbba6f19cf5054c8fd652557", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.2265625, -0.294921875, -3.03515625, 0.71484375], "student_probs": [0.30660977959632874, 0.18201543390750885, 0.011750045232474804, 0.499624639749527], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cfaf283d123aede98fa5bb555f43a54c4aa6bef1d12a0f030228b5e810252929:action", "state_id": "ffb666a261d695d4d670a14faff2d68b8d6baf541500ec502fc87c3f9188a162", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.3203125, -7.14453125, 3.81640625, -3.158203125], "student_probs": [0.000793969607912004, 1.7336715245619416e-05, 0.9982549548149109, 0.00093369948444888], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5bedf55880428db87db15e072b9b381885ea44591dbe1b536221ea35e4a300b9:action", "state_id": "921e9089bc5124e0a16f289406fb5b8c7af07dc25157be0843c9dbcc06e97596", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.693359375, -3.90625, 3.2578125, -1.22265625], "student_probs": [0.006941985804587603, 0.0007593421614728868, 0.9811837673187256, 0.01111495029181242], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0c76c3e3328a633a4377cdf73819928cebaf63f2e76b4ee317fb3c5ab6e84b86:action", "state_id": "c19e9c33925bbdc97155c8cb50e351fd9da09dff0308fab08ebbb58e513828ca", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.69921875, -5.3671875, 3.4140625, -2.390625], "student_probs": [0.002201432129368186, 0.00015276407066266984, 0.9946485161781311, 0.0029972700867801905], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ad481d828fb80aba7a096b6da8490228f7ec948e6424ccb9f599850e64e017ee:action", "state_id": "806ddd8bb5ee8dedd1b36cfd1965fc7e7710f0464c6a4f39643e255adc89c21b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.508544921875, -2.30859375, 3.82421875, -0.140625], "student_probs": [0.012696078978478909, 0.0020985451992601156, 0.9668630361557007, 0.018342358991503716], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b6a52297dc1d2723ecb41bfd9cd99cbbca46c6c2ca842b22c75d0e8d59bb222d:action", "state_id": "7258b6e961e27fe42d7e01a234b894111a8d8ed27a97f2575215bbe708af5eca", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6455078125, -2.58203125, 3.7109375, -0.265625], "student_probs": [0.012409140355885029, 0.001789452857337892, 0.9676579236984253, 0.01814356818795204], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "95ddfc1af3560b2afff553492ab7063da0a8124452fc2b9bbef37316dac0b657:action", "state_id": "44bc5603433957a4c90fd32a6c153e23658a35ba82742d184fcc6293cf15b8be", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.412109375, -0.267578125, 2.59375, 2.359375], "student_probs": [0.14235283434391022, 0.0265391543507576, 0.46403005719184875, 0.3670779764652252], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7ce79295549ad1569753b5f5cd33154d54fce9be7cb9106dcdb78e17f5fd7c8d:action", "state_id": "fb27c2ab786f4d0569c343d139e43ca2a5513f3031830e77adfbfdda86e873cb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.23828125, -1.3046875, -3.1640625, -0.1015625], "student_probs": [0.5104847550392151, 0.10911386460065842, 0.016996663063764572, 0.36340466141700745], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e3cf7f8cec5b9566b1b2bd23e532840eee9957cbdd9a24468761d223943e17e4:action", "state_id": "5179cfcd08e786113624e2dba467f4698eebae7ebc6cd65ac76ad4c02040f675", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.5234375, -6.5, -7.6953125, -1.638671875], "student_probs": [0.9845119714736938, 0.00011868392175529152, 3.591486893128604e-05, 0.015333449468016624], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0eff34356e348a42fe0207b75aed070d5512de0bc2d42f4443440cebbc763a75:action", "state_id": "2965681e534a4e435903c3468cb8daa626b04f6471d6b2ceaf61654b85fc02bf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.2734375, -5.00390625, -6.9375, -1.587890625], "student_probs": [0.9786353707313538, 0.0006762552657164633, 9.780511754797772e-05, 0.02059052512049675], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "30ecd91d23286dbdd2dc60f498864793a13f8b5231c814b474f8dff564c62e28:action", "state_id": "795f2ab26312c93f377d29d43f5f68d45a6316bb9b0181594e16393402f57a58", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.8984375, -3.4765625, -5.703125, -0.75439453125], "student_probs": [0.929729163646698, 0.004305500071495771, 0.0004645578737836331, 0.06550072878599167], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6fdcdc83c70721d5108021f48ca744ec4e5504151fd8445eb4e747195fe89b2e:action", "state_id": "fe5f11bf75dcd1334afc25f845364c25feb2f2a5c1eb3e880f6ecb002f27b057", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.00390625, -2.51953125, -4.7265625, 0.68359375], "student_probs": [0.326555460691452, 0.026184361428022385, 0.002880981657654047, 0.644379198551178], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1b45e21bf52b2609c6b223de92e0f209be100f9d7bfeafc0dca355cf8897c5ef:action", "state_id": "15003b6924304410b3fd5635dde3fdec4527428ccdc986505c9a49cd5b17c36a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0234375, -2.62109375, -4.05859375, -0.546875], "student_probs": [0.6048653721809387, 0.04296881705522537, 0.010205987840890884, 0.3419598639011383], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6e44022f2ac9c1b9ac72c50e6ff067fe87105fe78c653c51fad7e2fe7246e52c:action", "state_id": "10b54bb75d08b5afd825874ad366c8e84215b57819f8b77aaac67141a26159b1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.955078125, -5.796875, -7.265625, -0.267578125], "student_probs": [0.33350592851638794, 0.002632315270602703, 0.000605993380304426, 0.6632557511329651], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "12018ee5046d51d74502f0c1f03d03afc1941e87e995ad7ea0948c77f94dc855:action", "state_id": "71471b73455794c8bcf7ff461d646a7a79a34234b98f1c7109ac71c0ab7f1ddc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.181640625, -2.98046875, -5.1875, 0.26953125], "student_probs": [0.37911370396614075, 0.023080959916114807, 0.002539524342864752, 0.595265805721283], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8feb86fca23d1f3cadf9eb5ffb20cfce7815df63ec0a20e9b0a36dd06193765c:action", "state_id": "dc695904d06deefd9f6bc0de4ceff9d6c8f6d387cecb8b29585bd7a183235e82", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.7109375, -6.1640625, 3.1484375, -2.7578125], "student_probs": [0.002836952917277813, 8.977987454272807e-05, 0.9943662881851196, 0.0027070397045463324], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c39161d618546f9cfebc0f5bbe3e81974aada1f224e75c0967b592a0cc77dc6c:action", "state_id": "cbda1a5a958a9fea906399aa9af8e5f38b3a836ae9843d901fa547534fb6e94f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8125, -2.6171875, 2.140625, -0.9140625], "student_probs": [0.01785685308277607, 0.007986078038811684, 0.9303048253059387, 0.043852198868989944], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "aa94cce2098dab2cdab35029d450e8b43308eabd2b54b8553fcb8c8e14686961:action", "state_id": "026bb6d0522ca10bb997ee75ffeba48de36552ba2b1b16dc7aabc19cf5011cc4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.71484375, -2.5390625, 2.17578125, -1.009765625], "student_probs": [0.01908249780535698, 0.00836915336549282, 0.9339253306388855, 0.03862306475639343], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f3f216038dd81f1b560473d973e4d3017d4d254f6c5cb71ed858c91356052ccd:action", "state_id": "9ff65003f4ce581dcaba50a6e5db98c7e72a864b587b0429a34f4ec7d28afbce", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3984375, -2.23828125, 3.001953125, -0.849609375], "student_probs": [0.011813949793577194, 0.005101003218442202, 0.962632417678833, 0.020452583208680153], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1f2fea8a2a7b333378a7c4a12a5538ff830f6b50d650a47f8171482238462be3:action", "state_id": "e34f922e88a9ec27a2fe7c908a5d86b049cf5c3da536ef337c730e139ca14842", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.140625, -2.33203125, 3.2734375, -0.6220703125], "student_probs": [0.011683914810419083, 0.0035495003685355186, 0.9651423692703247, 0.01962428353726864], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5c190ed22a7bc682c2ef1e43448e3234fc22df957bc3a0975088495cfb4ec1ea:action", "state_id": "34e11b1bfe8f04c882fa96d804a57a8a7f9dedf8129d83e3bc1443bb2ad1f823", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.10546875, -3.375, 3.69921875, -0.841796875], "student_probs": [0.008033006452023983, 0.0008302964270114899, 0.9806801080703735, 0.010456572286784649], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "389f834f65d6acaf9f3b71be157842733dad7ad073db872fcea1326d0fce7bac:action", "state_id": "720f8f14e13c497433828c941a4ccf791d1dd7745f5d89d5c1f50bf20f3f8ac9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.77734375, -2.3671875, 3.8046875, -0.44873046875], "student_probs": [0.009969526901841164, 0.002033359371125698, 0.9741490483283997, 0.013848076574504375], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2972b92e2c7467aab61124ec327e2b5b8c91c70f290dcdcef0bcea86d747ad39:action", "state_id": "a406aadb2a57a9e679441679eeb86be6477fc6a91201eaecf791afff76e9a05c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4169921875, -1.984375, 3.9609375, -0.05078125], "student_probs": [0.01214716024696827, 0.002533780876547098, 0.9677996635437012, 0.017519356682896614], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8fb44a3f13feaee7d4826952015bd1bb52b32bf9354d931651109b4c975e8a41:action", "state_id": "3b1ba61584762ccfed521da84fd89445041c1aa19199cb25b4a2ae2f324a41bd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.73046875, -2.49609375, 2.6953125, -0.73828125], "student_probs": [0.0303859431296587, 0.00519842142239213, 0.9342660903930664, 0.03014947660267353], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cf3839263c2ee8fe0e96ff6bd3fe2f7ac2386abdc5f0db30be7a3e89d0ed6912:action", "state_id": "172752a69277ce3c49ce8e513596b3a15f47b0baad1d9fbcb13e65f44f0ea250", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.677734375, 0.2265625, 1.6640625, 1.47265625], "student_probs": [0.1530802696943283, 0.097493976354599, 0.41046497225761414, 0.338960736989975], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ed95251bd4213d771940b623362f3207cf5aa2067aa99509ac67c424d5d3ab77:action", "state_id": "dacb9b4c712bc60d3a1163c088c400babb77b258ee7429adb5304915cb7a11b5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.29296875, 0.28515625, -2.83984375, 0.78125], "student_probs": [0.27282702922821045, 0.27070388197898865, 0.011893898248672485, 0.4445752203464508], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "52b10317ac0fae39dc021672a42eac0c90fe09df34feed3d6b8c85c359de7f6b:action", "state_id": "e0a6f513772d7061466c724aefb7690a16f066b402a45ddb11f00514b2561cc8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.994140625, -5.7890625, -7.3984375, -0.2109375], "student_probs": [0.31265661120414734, 0.002586184535175562, 0.0005172694218344986, 0.68423992395401], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ac5c4a321f7c26588cd915bc05cd7bbbdbc1ca6d5aab65cf91beaad35013d723:action", "state_id": "1051b130d8ce84c11a45783cecb6ab90aae0e19de38cf359003f850654563813", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.16796875, -2.8984375, -5.1640625, 0.421875], "student_probs": [0.3477463126182556, 0.022669140249490738, 0.0023522668052464724, 0.6272323131561279], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "791eea1642c2e859e8cf6ddb69c999f1c3cc9467bf82106bc89e374dd14ee4d3:action", "state_id": "509ea2b34791308dcaa0ca41abd54188b5819b71876f38a38b71e3633c85b044", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.8203125, -6.5390625, -7.96875, -2.7890625], "student_probs": [0.9962440729141235, 8.583033923059702e-05, 2.0546385712805204e-05, 0.0036495989188551903], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8e152ebfe83f6f27e192c8b397491c339146fa3034579210894fab33f3cafade:action", "state_id": "e5592fcd2cee380cd57d09170257f175d37f47ffc0733144289a38a12e13db47", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.25390625, -5.625, -7.671875, -2.375], "student_probs": [0.9899099469184875, 0.0003748263989109546, 4.840426845476031e-05, 0.009666900150477886], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9020b26e5029c4f8d59f44241fc1ea1cca43d61614807c14bab2a57e57b6e477:action", "state_id": "cc60a2af528c55532130d8f55a09868f17665f8f7fc5825091865fccec2d6a67", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.3828125, -4.09375, -6.34375, -1.578125], "student_probs": [0.9796750545501709, 0.001507810316979885, 0.0001589220337336883, 0.01865815743803978], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bc0bd44b943dda85cf928ecf0c16d1cc838e15edcc709b36107719f5b7fbcd28:action", "state_id": "382796f9fc1bc03faec07173ef7c68de9dcb331c422f421f941c69686b59eb84", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.17578125, -2.8515625, -4.54296875, -0.70263671875], "student_probs": [0.9398603439331055, 0.006161914672702551, 0.0011353958398103714, 0.05284236744046211], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1a0bebf005df4cad3d5a1233d266a7abc476967edbc6fc10a42d1d99d5891434:action", "state_id": "82ab7ce76f19637977880f1dac39204f0709772a3d4f90e009020f9d10065eff", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.6953125, -2.23828125, -4.10546875, -0.162109375], "student_probs": [0.9383238554000854, 0.0067564756609499454, 0.0010442655766382813, 0.053875360637903214], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ca9ef4d414547622069bb346f7dee2e860336cd0107748c4d42b7ae811aeda7c:action", "state_id": "fa9521ed23953ea015d26a6433ce9a4d547c0636840cf28a8a057b1523788c46", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.609375, -2.03125, -3.640625, 0.30859375], "student_probs": [0.7669873833656311, 0.02012263610959053, 0.0040247803553938866, 0.2088652104139328], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b4bb20ac5b3f12a125ac0f7bc449f396c73e850a6d8f90c7caca3b84c29037a3:action", "state_id": "f198c14edc992d003a00182a7caffe5fde6fe89d17980b9b3183b8c6dcfd0338", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.41015625, -1.44140625, -3.4765625, 0.82421875], "student_probs": [0.371684193611145, 0.058351319283246994, 0.007624187506735325, 0.5623401999473572], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0eede9a8e2c8e1ee615a4f2d47f329c7e97abe12cfc8ba7d0c3a7536aa601122:action", "state_id": "5a455f5a07dc291abe2a0eeaca9d0644451c6711fe341dba4e1ea7f8a929c47c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.44921875, -1.41796875, -3.21875, 0.375], "student_probs": [0.47425854206085205, 0.0733003318309784, 0.012107000686228275, 0.44033417105674744], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c349d93df77f0a3cd2a21cf95a8beaa7723fae0d44d7861a2886921c994c94bf:action", "state_id": "39c03680674e25ba088d04536412bf549ac764eba827e32c706d65c5705ec785", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.37890625, -1.48046875, -3.21484375, 0.14453125], "student_probs": [0.5065008997917175, 0.078897625207901, 0.013926257379353046, 0.4006751775741577], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e5b468952dfc5d065c3b122e41e0e18b7e7abcba8253255349616fd8a2ace213:action", "state_id": "f74c76c3e8e899df91277e0da8937158ee7917920d31dfab66bb7eb90572d034", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.27734375, -6.2890625, -7.4609375, -0.85107421875], "student_probs": [0.9578210115432739, 0.0001823649654397741, 5.649402737617493e-05, 0.041940126568078995], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3977e76b8603c7dd9e25b887376ec4a0642b7431a074f7d3d0aef1f907bc3abf:action", "state_id": "29b85a79388a0c79b565ed42e2ea8304f43c24bc8077cf4e10821b15ae0863bb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.92578125, -3.22265625, -5.9609375, -0.423828125], "student_probs": [0.907778263092041, 0.005272806156426668, 0.0003410525678191334, 0.08660788089036942], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "64963ee641460fd90a2ec134e90b309eef9234f34a60056e8692999951aa9f54:action", "state_id": "960414ec02f2452ed772b6fdaf4f31fcd55a4d4530b55e8148891bc752803092", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.265625, -3.0859375, -5.0703125, -0.2734375], "student_probs": [0.48543769121170044, 0.028925931081175804, 0.003976346459239721, 0.48166000843048096], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4d2eb667d342f6ef990af0da0096d6673d1b5f236687bf0086e3eb5cef699ca8:action", "state_id": "fbb4232b76b7f4ba45e8214f7e784a249a7896793e60ea98c90a5696be44b3db", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.078125, -2.4453125, -4.34375, 0.0390625], "student_probs": [0.4480051100254059, 0.041997797787189484, 0.006291374564170837, 0.5037056803703308], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4648df67d499035b0082bff37e3ee4fcda87b4ce5baa0801c316b39dd5f57181:action", "state_id": "90aa6ab4002d5c2077cba41172040985e71ab0a26416fd7d7a5765cfa6a8ce55", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.134765625, -5.8359375, -7.4296875, -0.37890625], "student_probs": [0.3184337317943573, 0.0028928511310368776, 0.000587718328461051, 0.6780857443809509], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "37484f3d18100d58aefb42c0237faba2b06aca9033018eb1d1efd5292d67ca42:action", "state_id": "c87f7ae81361b2b4b225eb3e1569a198a87bcfc653634df4fad0b31ea257f940", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.298828125, -2.9453125, -5.2578125, 0.41796875], "student_probs": [0.31991979479789734, 0.022682324051856995, 0.0022458541207015514, 0.6551519632339478], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9d2944288c4d0d30c4e1c5bd169f8bb7a51bca0cb61edfb6bea1be2281a06566:action", "state_id": "673f961dd83abaad6f20f9c9a8711f09ffd990032332b69ec522893c577e2030", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.20703125, -5.9375, 3.0078125, -0.458984375], "student_probs": [0.05563778802752495, 0.00011935313523281366, 0.9156588315963745, 0.02858399599790573], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d6dec3214553bce4a94a384da012f45f4dc887f36dcb65023e9b077cf5c0cf5c:action", "state_id": "b8d6b542cb20d75e2c96943b05a526c211063e538a671f5f6ba601d6bfe3e2f1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.00390625, -2.4140625, 1.7421875, 0.32421875], "student_probs": [0.12179777026176453, 0.010937593877315521, 0.698165237903595, 0.16909946501255035], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "16ba93859370e74e8966d507de12c4b640d5216fd721a1809b5ee4a20e3484f7:action", "state_id": "b626b5606216a779bc6249ae16f6ee0038215365f0557c7aa9dcc0ccae231fb9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3837890625, -2.58984375, 1.59375, -0.47216796875], "student_probs": [0.10810238867998123, 0.011905781924724579, 0.7810333967208862, 0.09895844012498856], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9f891b966cad2380d9ef90fd86127344a63e5491079d6414f95dcee086a4a517:action", "state_id": "1a252fdc90f914ad2237bad7a8a51368b087d0b2bf34caef388ca4c736cbe472", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.10546875, -1.11328125, -3.734375, -0.0703125], "student_probs": [0.4119730591773987, 0.15037699043750763, 0.010935907252132893, 0.42671406269073486], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1858561f044f31538161ee511741d343c54798da9941650fa7d8ccb10cf9cade:action", "state_id": "c97dbefc325f678a06d2072afce427ee4ae37c3c15100d0a73ac8710648ff72b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.35546875, -7.25390625, 3.6875, -3.208984375], "student_probs": [0.0008718706085346639, 1.7675925846560858e-05, 0.9981010556221008, 0.00100941420532763], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "17720fb115bbb63ac5c724de4561929560fb13c1fea74e18ba36c78f808bad0a:action", "state_id": "af3a5dc7646a621f5834e2f14584d74a13777e9a397bceb1487f95d0ec31fa47", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.638671875, -4.0390625, 3.22265625, -1.232421875], "student_probs": [0.007587971165776253, 0.0006880963337607682, 0.9803330302238464, 0.01139089372009039], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d33740d7acf9dc9290fa298c2c76f89fced53ffa94b2c21debd9fb86fe442280:action", "state_id": "6404f92858fcecc4cb1ce292652c3b7996da12ce8e694c966a54c5170b2c3f74", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.798828125, -5.53125, 3.3984375, -2.48828125], "student_probs": [0.002024977235123515, 0.00013174810737837106, 0.9950809478759766, 0.0027624149806797504], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d4ea04df5a510c37e0c75b09bee0c9eb831e68a95531cf41bd21d6d38092f1d1:action", "state_id": "99c8200849b7b1d6f9b4e8ad1fbdb27dc0b0305b9430d61c137ffb6c867d4b4b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6865234375, -2.48046875, 3.8125, -0.298828125], "student_probs": [0.010802733711898327, 0.0017965245060622692, 0.97148197889328, 0.015918700024485588], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bc6043448f338b2fc86c5c34c691e928821496683c58c55507c59dc4c68ecb7b:action", "state_id": "85ad1463f55933f236cd663837e4447e66b3014a1ec43172a30124210e990b2f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.798828125, -2.7734375, 3.69921875, -0.3349609375], "student_probs": [0.010802574455738068, 0.0014995652018114924, 0.9705194234848022, 0.017178380861878395], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6c243456726ed9396ebd574d18f9d1253ca0d2ff2af20bf32e603377c1a1898e:action", "state_id": "6717e8a4d2c20eb0afbdd6dd419ad14f0470b8a0be79744dc0d79f74e94ec0fd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.482421875, -0.27734375, 2.43359375, 2.357421875], "student_probs": [0.1623455137014389, 0.027937259525060654, 0.42027056217193604, 0.3894466161727905], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "50b2c59d7575bb226d4b4513ce0ecfe2acf48eeceaff0987758f0c3a287ae2f6:action", "state_id": "1b744117a44370940e44463e536d5b088b3e2ba719f8256436572574facbb400", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3671875, -1.2109375, -2.91796875, 0.06640625], "student_probs": [0.5040231347084045, 0.10401105135679245, 0.018867971375584602, 0.3730979263782501], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fe9fd3b4c25359b8ce7a83d7a203b05651bec860fef073dc10e62487c71f8dea:action", "state_id": "bd3675c7c8af72b3a07fa297ee94c84c0910d96780961c513a545a95496174df", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.76953125, -6.3828125, -7.984375, -2.470703125], "student_probs": [0.9946029782295227, 0.00010539921640884131, 2.1246511096251197e-05, 0.00527041545137763], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "be40770881a79b87481898b04e081f836f90fd320fa5c3184186f7e2de7c81ef:action", "state_id": "3cdaaf680d0d985c70b420c58cb03e853977818258f93f433e552f104041f351", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.25390625, -5.3515625, -7.640625, -2.0703125], "student_probs": [0.9863955974578857, 0.0004909508279524744, 4.976348645868711e-05, 0.013063717633485794], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a8450cbd0218f5a45b7beb87e4b338475a504fba9ba4b5b18d1b1c963cd17d69:action", "state_id": "a862887c489015d9db40daa85942b7b224af505d974bc35b6dc2739582c580fd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.25390625, -3.8828125, -6.40625, -1.763671875], "student_probs": [0.9800732731819153, 0.002118924167007208, 0.00016990277799777687, 0.017637886106967926], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "28f8f64facbdd136f62b9b52478ab81e29ad55d6579a0c8f916cbda36c3a4cfe:action", "state_id": "bb137f053529833215a5f894d7a877b651dcb2c4f880531d2aac7bb515208617", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.80078125, -2.42578125, -5.34375, -0.537109375], "student_probs": [0.9604543447494507, 0.005159522406756878, 0.00027883786242455244, 0.03410744667053223], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f03f0455e5ba8c72ec99adc7de754f1976c3b83e469687353e26f0c582864e82:action", "state_id": "6571f9b46d6badc5686ca6896526711ccf6b979e792598d1fa2b511754582481", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [3.10546875, -2.3203125, -4.1875, -0.458984375], "student_probs": [0.9676846861839294, 0.004259386099874973, 0.0006583211361430585, 0.02739753946661949], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f41e808de4092c88772fb056742aae2b40cdb0efdcd0590f1853e1ac2919f5a3:action", "state_id": "9b666bb34bc177444d594a782f690963edfed459df0dbe6b8b1ceac05e3944eb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.8125, -1.578125, -3.48828125, 0.3203125], "student_probs": [0.9116129875183105, 0.011297602206468582, 0.001672692014835775, 0.07541665434837341], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b3470efac8407751858de05179669816caf6df08e0c048a36bcd40cf3c108f21:action", "state_id": "149bac9e30e5d3844a65385abc0274c74672788bb0d7efc86fb23f50701351fd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.06640625, -2.11328125, -4.265625, 0.0703125], "student_probs": [0.4365571439266205, 0.05637603998184204, 0.0065515427850186825, 0.5005152225494385], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "12d02ab395228c2bd2f2f3640249b0eb1f01567fc23248f28bb9d61348ae0411:action", "state_id": "708740e2e22c167706838aa893743d4015a04470ece435051ad19e20b7df53f4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.1171875, -1.83203125, -3.48828125, -0.4130859375], "student_probs": [0.5688269138336182, 0.08099257946014404, 0.015457702800631523, 0.3347228169441223], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "715e04f128107ee74e75b7eb535d98e591504e04c3ecabe02a3edc1dc50d88ab:action", "state_id": "aac40ee5b5c5a5ec4d4130b1cbc93ce4c75f3a2f3ca148cd0fd052a3a15a679c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.693359375, -7.75390625, 3.8046875, -3.57421875], "student_probs": [0.0015031612711027265, 9.533184311294463e-06, 0.997864305973053, 0.0006229500286281109], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "93f00e1ff3ae9dfd43c98fe84b5eb066ec5a23eeb87c222ffc4c530e682cb651:action", "state_id": "94046aff6e2df59d0fe5204e0f0c47276e4dd5744bc1d8b1b9827b1a453d54eb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9765625, -4.9609375, 3.2265625, -1.599609375], "student_probs": [0.005424522329121828, 0.0002743240911513567, 0.9863930344581604, 0.007908063940703869], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "11fe5d30bf6dfdd9a74962e8cf08eecb079a469b19c6ad7d31f3a5c7c9369ffd:action", "state_id": "81ca45709d96e3aeb8ec304128106c9da9b5543b07950742896a2ae677be3eef", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.39453125, -4.12109375, 3.25390625, -1.126953125], "student_probs": [0.009363827295601368, 0.0006128050736151636, 0.9777867794036865, 0.012236609123647213], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ceaecc82ce64853447ac4b39fe509d407f1329d6411ee41446e0fa8f0aa326f1:action", "state_id": "df7d6004954f79647c5230bd93688b40720b103cea0361a765d93af8a6293422", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37109375, -3.43359375, 3.1875, -1.07421875], "student_probs": [0.010212216526269913, 0.0012983375927433372, 0.9747474193572998, 0.013742038048803806], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7a6b2fd61f89ff80a7ed08158018166cd94d944cdc5f87dbf9c967d3de57fd74:action", "state_id": "a090ebea96465475ebdf66c612f09187a1301424015f5abda9bc1e288dbb74f3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.919921875, -3.0234375, 3.296875, -0.7373046875], "student_probs": [0.01425754651427269, 0.0017398010240867734, 0.9668885469436646, 0.01711411401629448], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "da3665c431cce761bcef688ea2bae43b50a1e1b7c4dcca81758465ee143f8130:action", "state_id": "f021af3be54c486870fa048875611a2b53a96a535ca8c65355b5fc1cb72a6276", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.48486328125, -3.43359375, 3.53515625, -0.614501953125], "student_probs": [0.017351163551211357, 0.000909308553673327, 0.9664979577064514, 0.015241485089063644], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2d097dacae19176f6a9644d255083ef3df2862f3060fdc7fd1e4e1ab7786a632:action", "state_id": "49dd7f2cdee8577ef20b5f0f0f62e9943bfa0f9fc25dd0e4ae18e7a87f7f7e66", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.568359375, -0.68011474609375, 2.453125, 2.021484375], "student_probs": [0.19603240489959717, 0.020693214610219002, 0.4748721420764923, 0.30840227007865906], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "aee804cc68f14f60af3215bf0b2398d755cae324076a48049fd09f7f7b6c8b47:action", "state_id": "bebe69c6b6da303e3469d3e4015f338ec643a9457e7357735558f81bc11780d6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.74609375, -0.1875, -2.7109375, 1.5078125], "student_probs": [0.2803777754306793, 0.11022725701332092, 0.008838407695293427, 0.600556492805481], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "65c63abdd0286153518385e583684ffc19aa9ef0c546240c395b091bc5e5f6a4:action", "state_id": "f55d621950978ba68a57b9cd2ed465c10766be82f89e3840b0b973edfb23dcbf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.5390625, -0.287109375, -2.82421875, 0.73828125], "student_probs": [0.37136176228523254, 0.1625531166791916, 0.012857090681791306, 0.45322805643081665], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dae2c8de33d42ef4724165fe5c7a682f669043063d0b2b49faf487ffe0eca663:action", "state_id": "8aec59b33beec67d4e4ff4efcc2a90c04d968e3397881e7e773248d69e83974f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5048828125, -6.1015625, 3.1796875, -0.93359375], "student_probs": [0.0241062231361866, 8.943799912231043e-05, 0.9601027965545654, 0.015701545402407646], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "04ddd6d7d802fe57dcee74f3545d51afa19632d67c7a1bedafe98bcbc08d57b2:action", "state_id": "b0294aaf259defce54f57eae1c710ea663317fc556fa0bde601c46b94478ecb5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.02734375, -3.01953125, 1.87890625, 0.6875], "student_probs": [0.10692480206489563, 0.0050796931609511375, 0.6810858249664307, 0.20690962672233582], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "243161fe7403458b2b041ac744630ed4f1064e40209e35cea76352fa51619862:action", "state_id": "5bc88338175e56b57e64942004ad66dc1d4ac811cc9bd6be797dca79ba552262", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.04296875, -2.578125, 1.91015625, 0.2109375], "student_probs": [0.10616853833198547, 0.008413786068558693, 0.7485610842704773, 0.13685666024684906], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7d48999e4e4c92faefe70e3eabe69144aa2dba5f160cfed658dfc012f9c98eed:action", "state_id": "ab59ba57f06bd106e310bfffec64cef71a02fd091bb5fd33ca5bc8822d62e988", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.255859375, -2.1796875, 1.2265625, -0.359375], "student_probs": [0.15500736236572266, 0.02263832837343216, 0.6825900673866272, 0.13976424932479858], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "36b3ce18e988a8e9e58042029a14f1abda4aba90ece3d50eb9e53ef3bf1159f8:action", "state_id": "92a1d4933a439674f064b5b8bbeb6b55579d43141dac3c78a6573534ef7c9970", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.05078125, -0.892578125, -3.46484375, 0.21484375], "student_probs": [0.3850100338459015, 0.14989124238491058, 0.011446046642959118, 0.45365267992019653], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7158a9480bbcf22d37efdd19f54223fdfd62e8079a116d205f58094339c82137:action", "state_id": "1cc063f8bbde6b87020e2e802a34cc91c1207cdf6518e80ea6515a272241317c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -5.78125, -7.4296875, -0.283203125], "student_probs": [0.31933629512786865, 0.0027736136689782143, 0.0005335051682777703, 0.6773566007614136], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5a5d29876c853dee54a9cb0edb1a1a55d5b8b300cffa9e5b9f720f8aca1ea11f:action", "state_id": "45358902bf3e5d38b6cea720f15fa3acc6d6ec7b22e30962ff38eea71959dac6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.216796875, -2.93359375, -5.2734375, 0.33984375], "student_probs": [0.3549555540084839, 0.02345762588083744, 0.002259971108287573, 0.6193268895149231], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2d90103a88855395f24bb142667d0e7a151e898c272417e21b75c86910241175:action", "state_id": "029646c8606ed24bbbaa69126570efe2b0c6b633763b821f00b32fffcd15e406", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.42578125, -5.9296875, 2.91015625, -0.22265625], "student_probs": [0.07397422194480896, 0.00012850953498855233, 0.8872188329696655, 0.03867831826210022], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "de40e518029ed44a83871c18e1d7e94921511f3f8a38403e02c608b59b38c139:action", "state_id": "1d91d137df7b881d681a463efa2c37cfcdf942f475f94d9deccb6c8307006815", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0390625, -2.40234375, 1.64453125, 0.3203125], "student_probs": [0.13528108596801758, 0.011774645186960697, 0.673725962638855, 0.1792183220386505], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "016769ec5a71c7fc56c09977d2ba6e8d0be0539ba88019056e18b30e9b9bcfca:action", "state_id": "7396f00f37526d5939863d08fa13cc1d18bfc299a3e3605e47a46f7d9510acd9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.74267578125, -2.171875, -4.3984375, -0.53271484375], "student_probs": [0.40016448497772217, 0.09583964943885803, 0.010340972803533077, 0.49365487694740295], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "18014cae2941f5ab497120f04a1abbc08ff93b801cd65fd6b973e306563be4d2:action", "state_id": "33f8192ffe4ddb82e1e9596d701f322ae3008f55b659bebb9483bbb1df9743f8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.451171875, -6.0703125, 3.1796875, -0.888671875], "student_probs": [0.025384243577718735, 9.208787378156558e-05, 0.9581343531608582, 0.01638929918408394], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c5def94fc5e4a882780f70cedb65607fef4cfd96bb433dba134f56c265023616:action", "state_id": "d8c50fe4cf579db9ef3ece4ade8576a6a9c270fa96211072b46f8fa2c7faf81d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.201171875, -2.46484375, 1.8828125, 0.21875], "student_probs": [0.09378895163536072, 0.009751052595674992, 0.753727912902832, 0.14273202419281006], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "514fd00f3823e71413c447a7173c4ec372fcf2a2df3d0535a24a089c61e61b67:action", "state_id": "d3a19de524e406050c91897bfbb570677ad4db67fe4d2a3867fd06d20e1b7d15", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4404296875, -2.625, 1.4296875, -0.4912109375], "student_probs": [0.11693075299263, 0.013157758861780167, 0.7587702870368958, 0.11114110797643661], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c2de9757ebf7836774c8f0d2a4a219b02a5c76d714d6a2f5ba0e883f8edc78d1:action", "state_id": "75ba6a10b1b0ca63a91eb8bb04a847843a61018ccef1d8cebbd297d1aaad9471", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.19921875, -0.921875, -3.51171875, 0.6015625], "student_probs": [0.3514070212841034, 0.11453167349100113, 0.008593513630330563, 0.5254678130149841], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "be933dae950d2b34958b9f2050c91b04a1bea520e720a8777a2e768b80e53a6f:action", "state_id": "f7fbda94ff43a25386a44e0e23670bf44761a54a22474edc040bdad906fdc96e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.359375, -0.9921875, -3.36328125, 0.015625], "student_probs": [0.5019799470901489, 0.12993024289608002, 0.012132695876061916, 0.3559570908546448], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6e110394b5e36f9f21c73a98b143de5ee0d0c80dc51ebca06c4edbeed1622194:action", "state_id": "a94848a4543021d3f7160923730c8e7f9b98c81f5da43ea5d8be5dcaef5219e3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.853515625, -7.78515625, 3.70703125, -3.57421875], "student_probs": [0.0014121269341558218, 1.0188011401623953e-05, 0.9978908896446228, 0.000686872866936028], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d53707bd815f58e98c120ed3dd8606066ef209bf8c7ca06e94aabcba7d308c09:action", "state_id": "8a08710aa31507ba00163c90e6c1a1be1f276665e1eea3d625e9579200c7dbfe", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.912109375, -4.984375, 3.2109375, -1.6015625], "student_probs": [0.0058734905906021595, 0.00027203719946555793, 0.9858419299125671, 0.008012445643544197], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3f48c87a8c7e8e3a35a1b73224df21fc44122086ee348353f0dc333bae7f6396:action", "state_id": "1775f59ab54812d0f9d7d5770dd52fff0468abf357caba7f7c83ed051f32674f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.533203125, -3.5625, 3.40625, -1.181640625], "student_probs": [0.007030047941952944, 0.000923944462556392, 0.9820543527603149, 0.009991712868213654], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e53073404e01b575a38f86998be84b79fb23fe6a109b15dea7b3e6f9201e414a:action", "state_id": "24f26ed33eda7d2e87484fdec045f3fdb149e322db341fb525f0613e42238081", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.203125, -3.0859375, 3.16796875, -0.88671875], "student_probs": [0.012246724218130112, 0.001863480661995709, 0.9690849781036377, 0.016804805025458336], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a68bf8c8933f45cfc862ca89126a12d4dd741494180bcdd6488b231f073f58de:action", "state_id": "ecf7c9b0024668d70fc50bbb022ae58f4ec840768f5ea2df49392f26545df289", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4453125, -3.25, 3.4140625, -0.33203125], "student_probs": [0.020154723897576332, 0.0012198782060295343, 0.9560532569885254, 0.022572217509150505], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "07284e080073047a7383a56951321ecce93f31cd55c6c9f7fb91e2bd5a7b40e9:action", "state_id": "5a2c4ca95946fde0c47acd00705546f84f32b16a9c22220b321ee28166259b7a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.552734375, -0.9306640625, 0.41796875, 2.33984375], "student_probs": [0.27762407064437866, 0.023170258849859238, 0.0892554372549057, 0.6099501848220825], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f3f66942343b9ce80c46ee15b45a144118b4799afce1f3829a39063942367a82:action", "state_id": "a42e02ee8fbf87aa4a4187e5e0d2c82fb17f6653a5a2d4e5c16907faab811472", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.8125, -0.3486328125, -2.546875, 1.3984375], "student_probs": [0.31800922751426697, 0.09957863390445709, 0.011053038761019707, 0.5713590979576111], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "51ef8571b5eba031a349cf6fe3826faea75acb3d956d67c7fbdabe08229a3a0c:action", "state_id": "a81e53b39f4970f8ef28840c62548d74b79b1be409bf49ae0e7a3e28cca4b971", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.4609375, -0.4501953125, -2.7734375, 0.4765625], "student_probs": [0.4069671332836151, 0.16362866759300232, 0.016028324142098427, 0.41337594389915466], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "df689138e02dc64eeece35446ba3f4168c6edcf39162632855342cee25fc93f3:action", "state_id": "ae4b1b33c822f7a9d6834e9cc42899c659bb0bddde20af77aaf9434716fe1031", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.73828125, -6.421875, -7.9921875, -2.552734375], "student_probs": [0.994862973690033, 0.00010460632620379329, 2.1756042769993655e-05, 0.00501076877117157], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5bff55c5b47b6fa8f89e0423b43286296a22cb1ad47536bde5869b6fb1f6d45c:action", "state_id": "fa98eba08bb8235a371bb45d0c4d7f9822e8608e598f075a26c7c9dbfaee6c0f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.21875, -5.18359375, -7.65625, -1.97265625], "student_probs": [0.9844589233398438, 0.0006003445014357567, 5.0645354349398986e-05, 0.014889942482113838], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7b41f43dfca2091fbb8e096c937f2c90c02c132d1759ac3c6677277a39c59528:action", "state_id": "256224bbc34c9577de743bd275f0491c29416cf3418d366f456a4baa9b96c874", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.1484375, -3.33984375, -6.140625, -1.34765625], "student_probs": [0.9664621353149414, 0.003996267914772034, 0.0002428235166007653, 0.029298853129148483], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d7721ca19dfbbc4f0d490244d5001431c2643c3af675033e71a87cb6a28eb6df:action", "state_id": "14474ae5011d01dbfb53bf051ad0624a0c8ff9f5762d538900220fc476470364", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.8359375, -2.140625, -5.109375, -0.1875], "student_probs": [0.9470721483230591, 0.00653265044093132, 0.00033556579728610814, 0.04605967551469803], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4e7742a3910e6fdda47ebdc455f23b607c0122d89544b139b7a47cfa6760d7a4:action", "state_id": "3442165877f1c641095cd33d77408de06823b3ea16e0024137e2dd929a499363", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.91015625, -2.171875, -4.48046875, -0.4990234375], "student_probs": [0.9616376757621765, 0.005969161633402109, 0.0005933401989750564, 0.031799737364053726], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2bef19dd5f51d7c4726215e51cc347f59c8acd3f51e1922d8fc8655fa263e8ed:action", "state_id": "1d03c2cff52cebecb87ebd799d020693b9b846f8cfad7cfb2254052dbcd600cc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.87109375, -1.708984375, -3.73828125, 0.03515625], "student_probs": [0.9343478083610535, 0.009580891579389572, 0.0012591964332386851, 0.05481211468577385], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "42cbc2477a2f44f00134ff0e9e4e7f7a8f5b898b62c45718b58300445489d289:action", "state_id": "b3c57a28ba87a6beb61f4dde786e50aba35e22012b1bb12f35f345e349c3b9c2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.51171875, -2.2734375, -4.6875, -0.12890625], "student_probs": [0.8206170797348022, 0.018632369115948677, 0.0016666869632899761, 0.15908388793468475], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "26cf24c220dfcd5df17df84a6cd764b346d8e040b5abd9ebc525d92073838412:action", "state_id": "0990f7d0853a4d29a89053cbdeb0de7e2865ad07fae868ac4522a2fd2feda04c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.4296875, -1.36328125, -3.33984375, 0.89453125], "student_probs": [0.35954493284225464, 0.05985173583030701, 0.008292138576507568, 0.5723112225532532], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0360295a9b58d1dc6e357a07bf7965d8ff66f60f33a3871fc3221f6ed92abbea:action", "state_id": "ec23ec1187e23a605965dbe7d884435182636afe7100b93aefd18f365f4335d7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.046875, -1.3515625, -3.09375, -0.166015625], "student_probs": [0.4532172381877899, 0.12293847650289536, 0.021531060338020325, 0.402313232421875], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8dee78bc0e73fd5f59c91f0921455ba2f21a8e4e71ebddb6c691605205236be7:action", "state_id": "b4a5f8573e4bf44b647d4f5b8be959705a3da94abd21478ae5c22f1d70f66445", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.47265625, -6.171875, -7.8203125, -0.33203125], "student_probs": [0.2415706068277359, 0.002198868663981557, 0.00042295290040783584, 0.7558075189590454], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "05b3b1c186ef6ebb4e93f8dde1ebdd9963b19ebaaad50aeed419d7680d822c4d:action", "state_id": "0cf2fe094f101376df5003e33d55506dc86f66c87d1f581b3d7b8bf67ed37062", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.635986328125, -3.19921875, -5.625, 0.625], "student_probs": [0.21678957343101501, 0.016704775393009186, 0.0014768530381843448, 0.7650287747383118], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1afae698e79e2f658ebd708dad8c1a18c8a8e96a9a5f4d9e11509b3de19dba41:action", "state_id": "a09f7a61a9621d635014115a9089a190b764e0a8642ac19417ece53c0bcc98ef", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.94921875, -5.7421875, -7.3359375, -0.20703125], "student_probs": [0.3214920461177826, 0.002664466854184866, 0.0005413193139247596, 0.6753022074699402], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f7f1682b9c1ebafbb3b5c1c7954fdabed3454f868615f90b1d3cee10a38d2b64:action", "state_id": "d88f78715a52aea569bff4697c06eee6da39b312a9786fda22c5f13e4e6716dc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.140625, -2.87890625, -5.1484375, 0.4296875], "student_probs": [0.352089524269104, 0.022773651406168938, 0.0023538987152278423, 0.6227828860282898], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "264c7c325241cc229b0eb951b149cb5b4b19ebb2bc221d1da7fba06af536e7a0:action", "state_id": "b400a1506e4e1bac966491d41c43d543b9f3b83836599fc6094fa04f3f0a1559", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.59326171875, -6.1015625, 3.1796875, -0.962890625], "student_probs": [0.022122308611869812, 8.966146560851485e-05, 0.9625016450881958, 0.015286309644579887], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6e2fb1c12abb9cde207e18c146f79bf2ecede95234154d4c0fd867dfff47cf5e:action", "state_id": "0fa5d00ce5dca5cdd2f800bdfb0b969b6ea03ae4ff92cbf20c41f5bb83fe4dbd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.236328125, -2.48828125, 1.83203125, 0.2109375], "student_probs": [0.09450852870941162, 0.009941689670085907, 0.7477356195449829, 0.14781413972377777], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bc7f45ebbc431643f3cbe05cd607d7511c5801a692bcc977f4050d390f649feb:action", "state_id": "9dae685a0340daafff6a9b6e637849b6193e75b062bba75011520aa324ac3af2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.44189453125, -2.62890625, 1.40234375, -0.49169921875], "student_probs": [0.11923288553953171, 0.013384092599153519, 0.7539430856704712, 0.11343998461961746], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c28eb657190561d9ab465d6053514e78fba061f57bf167ed43414550f38209c8:action", "state_id": "b504a082ae095829b5fca4489176454f4ab3fb3fe5884ea75d7ead1ba504b647", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.140625, -1.01171875, -3.55078125, 0.55078125], "student_probs": [0.3511376976966858, 0.11092282831668854, 0.008756289258599281, 0.5291832089424133], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b1a696fd80002f4a611185748bb2f57517694baee882118a36fd93f33eeffd28:action", "state_id": "e24b0b7d21b948a1745803446a8f277006bfd64573a45293519ba12657d395b8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.296875, -1.109375, -3.3515625, -0.0234375], "student_probs": [0.5007475018501282, 0.12271344661712646, 0.013035344891250134, 0.36350369453430176], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "91983bf5092b2db074758a53c4ca932c34208bcb576f8826fe16c68ed1f85f82:action", "state_id": "3696a26c85a47f3f1f2b136954dc4ecde7c664741d997c7f9fcc286e02897cd8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.515625, -7.640625, 3.87109375, -3.474609375], "student_probs": [0.0016798427095636725, 9.988709280150943e-06, 0.9976663589477539, 0.0006438533891923726], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a818e4caa8d7fbb9fb9f33d7064c3844eb9e597980a623cf44743b67afa3ca3b:action", "state_id": "e75cf25126dee9c70b5e1bdf63d1463adaec62eaaa35acfbbdb2a82b3f136490", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.671875, -4.51953125, 3.24609375, -1.396484375], "student_probs": [0.007189091760665178, 0.0004168239247519523, 0.9829257726669312, 0.009468358010053635], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4bb068aeef358ef0ea64ead4c8d154a98ba76b1b7ac57ab9a27e7c40f530b853:action", "state_id": "bda1ca6ca5f447f2b475862e585411e80aa0663f21e7bd1768c9bdf56a4a1b70", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.72265625, -5.609375, 3.30078125, -2.451171875], "student_probs": [0.0024075298570096493, 0.00013424114149529487, 0.9942997694015503, 0.003158463165163994], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "03f6e374ac95c2b0ef0a4d1b8c2e28ea7a76471b8842463e597ee3f6df06782b:action", "state_id": "ff0011c3f1eeffe7f4aeebc14f971bb0fd0e92e354eb3a97167b0b8829d9b00f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.77734375, -2.6328125, 3.70703125, -0.24609375], "student_probs": [0.010931473225355148, 0.0017094595823436975, 0.9687640070915222, 0.01859506219625473], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5540807dc215154a16f5ee228d71daedf0cf6edd6f735ae476ad3c7308d9fbb9:action", "state_id": "ca4c0e804539ef81cd215d80fb9089c423d18c6506a3bc90f9e8923a1594d9e9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.63671875, -2.63671875, 3.73828125, -0.26171875], "student_probs": [0.012190637178719044, 0.0016498233890160918, 0.9684222340583801, 0.01773727312684059], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7f3fa4d9cec3557f6dae214ee433646340396f86928be553a759826119b774e3:action", "state_id": "8d19cc14725fe64c7c8176c8776d354f00e103c8fa0b3a9822420457dc599a07", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.16015625, -0.439453125, -3.16015625, 1.63671875], "student_probs": [0.16769973933696747, 0.092071533203125, 0.006060926243662834, 0.7341678142547607], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "54a87031e18aca7c6e7161b4cbf37782bbada403e976ca95a6e880b62e06c872:action", "state_id": "2b66177b165c34df1bfbbad47dcf515b38412179ba8549a3575c9e39886a3403", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.28125, -1.27734375, -3.19140625, -0.0703125], "student_probs": [0.5141245126724243, 0.10818813741207123, 0.015955589711666107, 0.36173176765441895], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e03a34205b19953987a3764059547f7647fd1ebce6d4d37639760bfd2141a5b9:action", "state_id": "c0f24b741cc5b8d0fddc2718b28f0ca885140ed668197d5320b7a79f0778c3cd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.2734375, -6.0, 2.98828125, -0.5283203125], "student_probs": [0.060413192957639694, 0.00011392327724024653, 0.9123751521110535, 0.02709772251546383], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "347c22042a958f3055e14b7287611f00d19d03414e160df6a144c7f5e7c05614:action", "state_id": "175f48011019b072267daa2b6f553397902fbb30f7c1e2c90ae04266623fc6ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.765625, -2.359375, -4.6015625, -0.0234375], "student_probs": [0.30072757601737976, 0.06109651178121567, 0.006490030791610479, 0.6316859126091003], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "87970bf6c574a01c02846cf087d1712cd5e9b4dae3911c7f3aa244fdf91b5b75:action", "state_id": "a16c9d4a67b5d7ded80382008aaa83344f8a43eed9e853d8c52bda4d73695be7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.659423828125, -2.12109375, -4.3125, -0.54638671875], "student_probs": [0.4206216335296631, 0.0975206047296524, 0.010898851789534092, 0.4709588587284088], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2d50a6c935319136c634e645319afb5f8fe46f08314c83957a07f38bf86a8c85:action", "state_id": "61b484a07f8a3f3a50426ff7b8559ad43b43b86f1c807bc8c7a90c6032788f70", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4140625, -6.125, -7.65625, -0.583984375], "student_probs": [0.30262312293052673, 0.002722500590607524, 0.0005887820152565837, 0.694065511226654], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a66dc5e63762d8bee6eb2abaa30e2444a9225fdaf1d9ad2f3776aec9d8a99e72:action", "state_id": "ea1fe5bd53398475aed065a5e3d1dad69dd5420d5d48ac480f0fd0e4380778cd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.56787109375, -3.15234375, -5.5, 0.4453125], "student_probs": [0.2606200873851776, 0.019660096615552902, 0.0018793665803968906, 0.7178404331207275], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "739b9fa2dd4a989dd0fdd8571563801bbfb195bf697d4ca9a3eac2fadb8869ba:action", "state_id": "d469ce6a381fb68a8e4d1cf4ec2da8d2ad0b4a255248a1cfd2c5b19cf7f048a1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.935546875, -7.57421875, 3.83984375, -3.640625], "student_probs": [0.0004195260116830468, 1.1028178050764836e-05, 0.9990060925483704, 0.0005634324625134468], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2ca1c5624f5ac84683e93d6e7e4c374cd06c650ea017c6131a14c94ea10b0c92:action", "state_id": "c2d82098cdf2ac3c78a1b69a511154b588d71200fb1053b77364e0ee7244f652", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.955078125, -4.35546875, 3.296875, -1.4765625], "student_probs": [0.005164138041436672, 0.0004682970466092229, 0.9860344529151917, 0.008333252742886543], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "799071ff53af83f0bb4acaaf9d5472707cda757f99dc5cde72145af1cf821a59:action", "state_id": "6b69f6cf44decd530f588b0f2c76ee7f9868c0183084f5fc0a8462f23e378012", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.6953125, -5.9921875, 3.328125, -3.044921875], "student_probs": [0.0008883709087967873, 8.934581273933873e-05, 0.9973198771476746, 0.001702375477179885], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a9c65bfa61ee553eccf5a25d971e5ba42033153f4b101f2e06dccdef811cffa2:action", "state_id": "3a92314ba03b6df70f89e83419dcd9ea11a3224f3094ff253a9abc29c4340809", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2578125, -2.890625, 3.5, -0.953125], "student_probs": [0.008400368504226208, 0.001641258131712675, 0.9785658717155457, 0.011392589658498764], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "eb38e12eb1123be48e942fbae207affb094f3f119b229f0f970f644c3d730e58:action", "state_id": "7bfa87d525c8b5a92736b1376ec0aae1805c4b352aeb26c4480a338a19e602a9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9765625, -3.1875, 3.515625, -0.798828125], "student_probs": [0.01091456227004528, 0.001196212600916624, 0.9748517274856567, 0.01303753163665533], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cb46398a52cc766f08bb6d878feaffe6f5bf71a391c196fb239bee3529322227:action", "state_id": "b6c529fc95cfd529e4cedfaf1528c3bda1a80a7246bf0d96078831d3b929ee12", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.85546875, -2.79296875, 3.15234375, -0.91796875], "student_probs": [0.01751011610031128, 0.0025225712452083826, 0.9635180830955505, 0.016449229791760445], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "68263a9fd33f843350224abce8deddbf6ae52d3d061446b94c8c3d68326205ee:action", "state_id": "c521d3226f6d2a509fb265c762d3b961974c84a68ac673fb8d35cc49fa30db2b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1328125, -2.94921875, 3.6953125, -0.67138671875], "student_probs": [0.007829299196600914, 0.0012731151655316353, 0.9784777164459229, 0.012419884093105793], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dca809c10fff6781dbb0d3112ffde151711b33bd93c5cf7ab887e543e20c78fb:action", "state_id": "3a8832778ba711c92bbd3efcb945a5d264d40cd53894ed87432b2710c5385317", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.33984375, -1.94140625, 3.544921875, 0.15234375], "student_probs": [0.019420113414525986, 0.003914731554687023, 0.9448959827423096, 0.03176918253302574], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e3380b7eb4f385c5318911f54796c1e0fccc05a3558a453f81c2fb8175c319a8:action", "state_id": "86dbd311e0868b616825891c2d6e6cbfd19bd10f511818979a81fb854ef17954", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.84375, -0.5029296875, 0.87109375, 1.46484375], "student_probs": [0.2410343587398529, 0.06269362568855286, 0.24771609902381897, 0.4485558867454529], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "14ac2fc659e4ffafcf2db5e2af7b6e8e0ded4820638cbde87787a98d03363b9f:action", "state_id": "91951e5c0bfa0bc006171c8be3466b27818ecc59cfd014354d79c82570ae94c4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3984375, 0.17578125, -2.66796875, 0.7890625], "student_probs": [0.30076250433921814, 0.2407272756099701, 0.014012008905410767, 0.4444981813430786], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b1f24be5f92a6c6b0b3bd650c296e82eebaa5b40949ce3808f5b2468efb1532c:action", "state_id": "6b8dcc5c78368dc42284e40eccf88a01773c92c6ae0ac260997c22270ee4fcaf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.73828125, -6.3671875, -7.9375, -2.306640625], "student_probs": [0.9934667944908142, 0.00011033124610548839, 2.2946713215787895e-05, 0.006399877369403839], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a92989d8aa9b06fd955a92d755867198e094ab91441b6f0c9e79fc3b77a41450:action", "state_id": "34eb5382151cd0dc526ab60421369bd90ccddad4e6f47055f1450c67a5986961", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.20703125, -4.78515625, -7.4765625, -1.724609375], "student_probs": [0.9798226952552795, 0.0009004903258755803, 6.10402348684147e-05, 0.019215764477849007], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0e545a499e148c91101dc3fb3de6280bd4d22d60a8eb7fbbc2d7b68f818fbaa1:action", "state_id": "79db578c5cb88fad3113594f492da440615017a812d4f39ddedfaf009b802b30", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.22265625, -3.70703125, -6.3515625, -1.556640625], "student_probs": [0.9749563336372375, 0.002592714037746191, 0.00018418287800159305, 0.022266779094934464], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "631ea1fb040404f3c86bc5cba51ad36777d6dff744bf578a42c7074e05d7b69f:action", "state_id": "27ff2f35f4e11840e6fea0edf7076d835a97f4133a4be6fc5f4a658e78d0ef22", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.78515625, -2.375, -5.34375, -0.5048828125], "student_probs": [0.9585080146789551, 0.005502605345100164, 0.0002826549462042749, 0.035706717520952225], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "72ee7c0629b718f0c71c5eebdcf8560b31a695537a7b7bad464d12b91425be6f:action", "state_id": "8b14262b36058a5ed6b11a3e0465f746eef3b34e2e7a29039de8dfa1a22e2037", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [3.08984375, -2.3125, -4.16015625, -0.4208984375], "student_probs": [0.9660988450050354, 0.004353248979896307, 0.000686098646838218, 0.028861945495009422], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a3b0f29258b00bb9f5addab295d00c5861be7ab70f729b0b0c260df56023c2b5:action", "state_id": "25d9ec71901cc9cde43ebadf49a186f18a26fc7ca50afe12baf16cf026d04737", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.81640625, -1.572265625, -3.453125, 0.3984375], "student_probs": [0.9062792062759399, 0.011253459379076958, 0.001715691527351737, 0.08075167238712311], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b1200c25fd8f9250edd09137f97634b3fcd45e835e964dfb9c3dd27fa691b36a:action", "state_id": "e496de1313faeec203d273480b3df103cd50e78464f4fc2e58a45616c2867fe3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.5859375, -1.859375, -4.05859375, 0.25390625], "student_probs": [0.7696024179458618, 0.02454630844295025, 0.0027219343464821577, 0.20312930643558502], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f2c1bcccf7a0abeed2ca255e66301bfb5f1010cffcf41d4242b334e037444421:action", "state_id": "7911a9ff261191d5e0107cf3e92f129d26dec332df4b1d9d5b99d229efa66d45", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.21875, -1.69140625, -3.26171875, -0.263671875], "student_probs": [0.556750476360321, 0.08243094384670258, 0.01714400202035904, 0.34367460012435913], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b28a964117546ea9bb6d7d64cfcf028f43b22d880c314000cfa024e6c05c98d4:action", "state_id": "45ee2fde66949fe8ce7dfe1e77624d3a67eb7099599384c540b765dad92ee7d5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.16015625, -5.8203125, -7.4609375, -0.271484375], "student_probs": [0.29042771458625793, 0.0027488935738801956, 0.0005328973056748509, 0.7062904834747314], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d34375035f8fa4820ac44658892d1ca3a5dcc9177497f57af1e0e003eca9f72f:action", "state_id": "1eb17bb9a8540c4ec919c46f271fe4ac4f94333ea1bac4143eea30fcefd1e34e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.306640625, -2.90234375, -5.2734375, 0.515625], "student_probs": [0.29786649346351624, 0.022218875586986542, 0.002074766205623746, 0.6778398156166077], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bd415806c58c69f4c024b9324421fd8990ae582fc7888f7522f9ef205d0f272d:action", "state_id": "04077637e732c9e8889eec8709709995d5e32dd240c4428c1adca5d5f3a869fa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.70703125, -6.765625, -7.9609375, -2.45703125], "student_probs": [0.9942150712013245, 7.648178143426776e-05, 2.3144106307881884e-05, 0.005685340613126755], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "036444797bfbccd956f6ec8daedf21f0c9c8e19aa2408aae4111164e17554fe8:action", "state_id": "4edd9fcb7e1c415113da0659c0e9c4043f53e4769be45150ec472b27195cda06", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.3359375, -6.0390625, -7.546875, -2.361328125], "student_probs": [0.990685760974884, 0.00022841237660031766, 5.0569073209771886e-05, 0.009035233408212662], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1f6117fc13f26079bc6811fa071e61f034a2f62734e8cc4a68795cfaea0026cd:action", "state_id": "142ff1fcf1e764791d0be41a0b3c1a74e13aa55fb9e203dddb806bc392734cab", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.37890625, -3.7890625, -5.84375, -1.2734375], "student_probs": [0.9724843502044678, 0.0020378294866532087, 0.00026111293118447065, 0.02521679550409317], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "45dd0bf89e23341094aedc729f23a8edb15ad85d8ad15fc01a0b458bc6c8f0fe:action", "state_id": "10a953f384be6b3c37f870d849c62625688476091c59270f2d54ed3aa78f4ca6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.4140625, -2.58984375, -3.91796875, -0.177734375], "student_probs": [0.923041045665741, 0.006195154506713152, 0.0016415525460615754, 0.06912226229906082], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e3adf42169038fb6826c7d16efa6103af36dd9d74e2e0687cf73b997dea53ca9:action", "state_id": "ffffb5aeb12c0315fcf037c430fac0ee98a09bea00455c1772f67ec24cb40275", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.23828125, -2.44140625, -4.02734375, 0.3828125], "student_probs": [0.8564855456352234, 0.007949825376272202, 0.0016277723480015993, 0.13393688201904297], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f08a790b16afbe778a15b81935f805aeb5ffda6635a0374c56439a62e15538fd:action", "state_id": "d193dafeab0660852231d4c0c77a65dad486da66b295cc07d265cb62a78f53aa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3671875, -2.09765625, -3.5859375, -0.298828125], "student_probs": [0.6180590987205505, 0.05254869908094406, 0.011863413266837597, 0.31752878427505493], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2d71018f220a27a63af1c0b745eab816edf44a6a2405a2b987390ed62f8eafa5:action", "state_id": "32bc2731a20dc05d9939bef1cc4abe5859d35c76b53d7e1ae8f1542cec4c8273", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.58203125, -7.6640625, 3.80859375, -3.4921875], "student_probs": [0.0016732544172555208, 1.0386371286585927e-05, 0.9976429343223572, 0.0006734201451763511], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7d658eb8be47bbd38cddb5852a5acee9275d49b63244118a984f98ce6479b9b2:action", "state_id": "652da2ecc8557cbf79302dec4cc7595c9e896f5372bbd4c5e0e1ea06dd494891", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.65234375, -4.18359375, 3.203125, -1.388671875], "student_probs": [0.007643966469913721, 0.0006081502069719136, 0.9817978143692017, 0.00995015911757946], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "42d8c1f5f0ea6d0475fb504224b1e0576442baeda7f323342149e10993009109:action", "state_id": "9222f382f3e991c39ea9f05a13e701711600fb750fd73b45f20fe640e17bf600", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.13671875, -4.63671875, 3.10546875, -1.830078125], "student_probs": [0.005221271887421608, 0.00042858810047619045, 0.9872552156448364, 0.007094938773661852], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "68a4026b7b1e211848960840e4aaad03cf626648c7a872749a4f56d007282305:action", "state_id": "42c0f3a24d286c6b4093e81403598eeab6b9653c4e30fc621f1659782535bbe8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.50927734375, -2.390625, 3.55859375, -0.0625], "student_probs": [0.01635374128818512, 0.0024920585565268993, 0.9555889368057251, 0.0255652517080307], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "318d3f0ae09d0f83f3830b41cdb859a36933296c0809a67427124482cb86157a:action", "state_id": "1d693267d80da2e8e6076eb9953aebc436b6ae755ddddbec749560d2716e713e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.150390625, -1.052734375, 2.37109375, 1.908203125], "student_probs": [0.15074701607227325, 0.016651127487421036, 0.5109675526618958, 0.3216343820095062], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "38e4a17bcdc93a27181923ce0bf5104a8ba17057832cdecf6a1199397a2c4c12:action", "state_id": "0d543a1896ac217a38db3243715acf0df2355a7dbe87a914557072143ddf8c7a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.22265625, -1.24609375, -3.0625, -0.05078125], "student_probs": [0.49299755692481995, 0.11349448561668396, 0.018455233424901962, 0.3750527799129486], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3c439a27a9fe97e68209c8468aa419d662ee623de197db4933fa64ce429337e7:action", "state_id": "8d05a374e54a041aad84f5cbe134043cb506db58a15aff44310afb30187ea41f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.265625, -0.7275390625, -3.2421875, 0.58203125], "student_probs": [0.36067840456962585, 0.13359631597995758, 0.010806784965097904, 0.49491843581199646], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c23bffd2eab5b59ae6955e4928684427733f1282c604a636864efbfe7c33777d:action", "state_id": "75a2ca84870ba1e575d7f0fd73550c1128ee6b251decb8c15f4f7e5d0fae3f36", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.30859375, -0.4404296875, -3.01953125, 0.48828125], "student_probs": [0.3696131408214569, 0.17476347088813782, 0.013254431076347828, 0.4423690140247345], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6bd871b85e572312f9b42ffe63c1b6fd4f3e13d8fd3aca634cc8486e016a0a8e:action", "state_id": "b2569045207573ba2998a9a6c258ea1e4599a709526c746d920935a1cc6a2214", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.904296875, -7.578125, 3.7421875, -3.59375], "student_probs": [0.00047717595589347184, 1.2110310308344197e-05, 0.9988597631454468, 0.0006509495433419943], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8334da0b4eb3c1142928d7a10d27ca64681a6c3296693e606cc33af6b0a3386a:action", "state_id": "e9a8a5b64a56891808c2a90433a97170d8117a29b9b6e1c3f337ba93f28ac817", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0234375, -4.34765625, 3.34375, -1.48046875], "student_probs": [0.004606630653142929, 0.0004508043057285249, 0.9870140552520752, 0.007928512990474701], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "131c52cafe7d2ecb2931bd8841f76035ecf47f4ee8c19631f53c0018de111312:action", "state_id": "8c5c19249af3ecfc009dbf0056dff105757e51a8b9bdaf58766455db534260a7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.671875, -5.8671875, 3.34765625, -3.0546875], "student_probs": [0.000891879724804312, 9.928740473696962e-05, 0.9973554611206055, 0.0016532838344573975], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "843b3b2ead05cc5efa84f5511a2889062b36191dbcf4eb2aad8b8f79ef3941b9:action", "state_id": "8f3b01f2987180b545ca2e8e9f9ade9375ea6241d6d90a8ac6479bd57d6e4322", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.80859375, -2.5703125, 3.8359375, -0.5703125], "student_probs": [0.009393604472279549, 0.001613345928490162, 0.9770719408988953, 0.011921104043722153], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ab3129a7bcc15cc7b7c7fc459d0e45fd5c0fd694a012d9fc5fdbec57a3fff72f:action", "state_id": "ef942c3f4cd0031a96076f83474c272f979b34da779313c5cd30f67d08943801", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.71875, -2.47265625, 3.91015625, -0.27734375], "student_probs": [0.009512034244835377, 0.0016464994987472892, 0.9740513563156128, 0.014790188521146774], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f4f46bb84e927c6412fe94391248209d8b71027375b5243351b5fa1664bf9afb:action", "state_id": "32ad02021da74657042f9995d2bd347ebcd2604d4e761281f1e25e77e5a2643a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7666015625, -2.53515625, 3.71875, -0.5413818359375], "student_probs": [0.010973176918923855, 0.0018717973725870252, 0.9734100103378296, 0.013744978234171867], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "433f89aeb529c7c9439b4ea2489c1f19c6b2c96027b1356c5a169ee70644271e:action", "state_id": "17948cb77ad87200d8950f8df8e4dd85e5556b37f7f93190ba92c7ba735d61ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.58203125, -0.17578125, 2.36328125, 2.27734375], "student_probs": [0.1865338236093521, 0.032162465155124664, 0.4074273407459259, 0.3738763630390167], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dda56b4ace6e5304826b9a4c5ed459a23109494e9ec4b681706bf852e7063720:action", "state_id": "d861ad8c9f679de10a00f2da4785088c92b87e8a41c6950d002fe46ef2cf86a1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.4453125, -0.7568359375, -2.6328125, 0.2578125], "student_probs": [0.4596385657787323, 0.1381433606147766, 0.021164292469620705, 0.38105374574661255], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f020e9fe2fca4df9ef2a4cf555c75247c10a522ecb5197e98ed51e8a91d57a5a:action", "state_id": "01098b862b17480070304047f6910c03c63c2d68b6f8566659233fe610de2975", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.3046875, -7.18359375, 3.78515625, -3.158203125], "student_probs": [0.0008320168126374483, 1.7200634829350747e-05, 0.9981874823570251, 0.0009632731089368463], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "745c5b4b2ad1731a2b8f4fc4537e7d5f3ab9fde068e1480c53f67c9a628a42a6:action", "state_id": "7f1e17fc8de14bb4e0a1dd0a6262e676a62e466b5adb2c036468ae33a3dea20f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.794921875, -3.74609375, 3.2734375, -1.2265625], "student_probs": [0.006179672200232744, 0.000878177466802299, 0.9820327162742615, 0.010909398086369038], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ef5e3d0e0c71c49e636bf2d417257c479e890eaada4ca625538758c1450cc43c:action", "state_id": "2ec2e7191cf5c08fed546012bdbbf1531f119f345aea8f9896390d9a98feee4d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.62109375, -5.2734375, 3.4453125, -2.294921875], "student_probs": [0.002306354697793722, 0.00016256528033409268, 0.9943352341651917, 0.0031958082690835], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e0aa062bca57968b4be8037b547f13dda8dfa2606ee5ee2126346076718f1ddc:action", "state_id": "7bda4ab547fbdd50c0b8412de4210f9de6b840895e3e9fb19b98373c87276a74", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7236328125, -2.578125, 3.32421875, -0.5489501953125], "student_probs": [0.016772424802184105, 0.0026254281401634216, 0.9606284499168396, 0.019973747432231903], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e085ef02ea90eb164e7d94a306f156d4b38defbc100e08f625f6655dc5766374:action", "state_id": "aa12679e9916ba72d94006ce1003530de750a2d9c24897da2ef486c5dc7fe55b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7568359375, -2.61328125, 3.7734375, -0.45654296875], "student_probs": [0.010494234040379524, 0.0016394826816394925, 0.9736962914466858, 0.014169885776937008], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4066ce76183579306cab83703a3264f90d91ec612cc5ca9d4db2d9e25a59bf5f:action", "state_id": "c7901cfd4cb7aeaf1a5d31df6573f00100027f553f01b85af51969cef72d961c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.599609375, -2.37109375, 3.66015625, -0.18359375], "student_probs": [0.013609259389340878, 0.0023146674502640963, 0.9634456038475037, 0.020630406215786934], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f3ed55cd1f065e777c00718bf6bc0b21e55df2920706dde5672bc1b4d6229d03:action", "state_id": "6e2dd7e3003ad9dd0f5f7f20fce830c07409881f8cd679f4fa89c0a20325864c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.001953125, -0.353515625, 2.412109375, 2.1640625], "student_probs": [0.2646979093551636, 0.025106342509388924, 0.3989138603210449, 0.31128180027008057], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "79fa594e2f2df76e02f13bb4120dd7cb3f2d8e07445bfab7ec28e6d352031f66:action", "state_id": "31ce9b9d8ca9cdb085abc7d8691386b33e47ecbf7f9b349eb5a07894b7c0069c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.52734375, -0.518310546875, -2.65234375, 0.49609375], "student_probs": [0.4233173727989197, 0.14877986907958984, 0.017609434202313423, 0.41029325127601624], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6733e282f55d590589e1be18f2ea69a945dcdb6925f0cc5e232530095636c4fe:action", "state_id": "8cec9acb545da51685918abc634fc9f344a2bb444109a95346b46af0a5214ea9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.7890625, -7.6328125, 3.83984375, -3.541015625], "student_probs": [0.0013190287863835692, 1.039059497998096e-05, 0.9980486631393433, 0.000621849438175559], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6556e9ebbd1aaaf6b8cb11faf87b79c1efb60ba068fff50981702eaa29cf0907:action", "state_id": "fae0bcb1733a2ee9121c2068649d80506531c9361de6a99ae8380c9f9a7388ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.767578125, -4.3984375, 3.04296875, -1.40625], "student_probs": [0.007980464957654476, 0.0005747254472225904, 0.97999107837677, 0.011453836224973202], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "55cde20efe03e73740d471776e4e26d8e92c99e20f964bf95733056b087b6e99:action", "state_id": "a0ed58d1d48877338422f25975bbd983d591accab34eee76b6045228e73a6d99", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8515625, -4.60546875, 2.8828125, -1.77734375], "student_probs": [0.008625668473541737, 0.0005492707714438438, 0.9815348386764526, 0.009290210902690887], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d558ea33f1c3fa7e259da6a63f921b41913180de00cf31c7ca0d6ee189a3f5f7:action", "state_id": "fba6f090106fa9b700efd4e572431a0b425f752334e202d807cb376012e36ba0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.69921875, -1.146484375, 2.525390625, 0.94140625], "student_probs": [0.11571373790502548, 0.018272846937179565, 0.7185901999473572, 0.1474231332540512], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3c38935f17b4035bd2009218a4c9bda3b9b307e269bbacaa36adcfce73059528:action", "state_id": "fe353286ff7f9cf414ad6b61e0b5a3ac27069169519a04990fb17bf0afc8cf32", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.23828125, -1.015625, -3.28125, 1.3984375], "student_probs": [0.22195424139499664, 0.06334304064512253, 0.006572800688445568, 0.7081298828125], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a145c643f5b6f0fae42b3471c910e56d009395fc730c8f8aa3fe47aca2b8dfe0:action", "state_id": "90220bf826989b3c044f5f995c30fc942c4ef5e476f4e408e83bd98929e7018d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.21484375, -1.2578125, -3.34765625, -0.185546875], "student_probs": [0.5187417268753052, 0.11895554512739182, 0.014715570956468582, 0.3475871682167053], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6011b366c69a3fde6b5a74614ae12aca2ca443d1c56e2ba1eb3921ed99cbe9b2:action", "state_id": "280860a7e8a677c6552b1d61f1c0e48b3eb4ad7ebb2374ab9e21985f8b093c11", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.7734375, -7.7421875, 3.7734375, -3.54296875], "student_probs": [0.0014315721346065402, 9.952050277206581e-06, 0.9978952407836914, 0.0006631474243476987], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a07124d4af3507eabf963896cfa65c20d810991d1f479c4d097df539028a6d9d:action", "state_id": "e2e9fc2535c52320407488eb75fc4bf62137f13ad1e7ce4bbe5552e4599f9d8a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.720703125, -4.33203125, 3.1875, -1.466796875], "student_probs": [0.007259085308760405, 0.0005330850253812969, 0.9828504323959351, 0.009357331320643425], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d6f3d465dacc0a12ebe30e5269c7ffad6ed9e74f3448e3a8c0ec096a1dc7620c:action", "state_id": "7c6ca19ba6c2088e83a077e937f64025a0a20a5a23e2304213e1e5b7b114e40d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9140625, -4.18359375, 3.09375, -1.615234375], "student_probs": [0.00657770037651062, 0.0006798752001486719, 0.9838738441467285, 0.008868567645549774], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4c6a634781efdd4f9ede4491512ec04d01330e029f321c842a0debb97b837302:action", "state_id": "9cb32c8846d1f13f8ad25a2b014b3be0f92a1e23f62ce963e8c5f3a2cfd8f55c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.53033447265625, -2.4296875, 3.5859375, -0.09765625], "student_probs": [0.015619819983839989, 0.002337746787816286, 0.9579662680625916, 0.024076079949736595], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "907eaca04a852664ade482cb18ca36cb83711afbcc3c5480ac4ce08abc98431e:action", "state_id": "52b80f7b78fcfa536a561b1ad85733291ef0974a515eb87d65f66a432e8db5c4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.201171875, -1.052734375, 2.37109375, 1.92578125], "student_probs": [0.15647850930690765, 0.01642841473221779, 0.5041332244873047, 0.32295987010002136], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f62c98b39d466f03515fae7225ac20b5b8fc1fc74d02e07ecbaf1ee370c46d4a:action", "state_id": "7276bc2c098fa2e0e80f683e6c396f04d9960cf82b6de33d7bbabceff6e68b31", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.27734375, -1.17578125, -3.0234375, -0.03515625], "student_probs": [0.4994235634803772, 0.11678440123796463, 0.018405938521027565, 0.36538609862327576], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d9f768f8605f4aafb17cc86804f5cbf44667d8784ec29052e806b82b1dd0ac00:action", "state_id": "5732be6cfae925bbb5e13ff26e5719811728cf76d2af4096e9f86d9ab97b47fb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.25390625, -6.34375, -7.625, -0.9072265625], "student_probs": [0.959128201007843, 0.00017699541058391333, 4.914985038340092e-05, 0.04064565524458885], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "59f2726b7f52f81dc00bd89a9d25971ba99db7aab7ac3cbfcd5fca00d8834e87:action", "state_id": "a86bac1a85cffd191202cba9eb6a739b24932871896620e9bbaa2dfec2b99d4f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.021484375, -4.01953125, -6.484375, -0.556640625], "student_probs": [0.3779444694519043, 0.018853535875678062, 0.0016029677353799343, 0.6015989780426025], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d3fd84d76d66412b658769e57720a6beaa0a1f57fdc91940aa98ac1938d5cf64:action", "state_id": "7b636bad667bb63046b7cbf2e822954346c9e33d2e27614f12e6ec07c9d6fb82", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.656982421875, -3.54296875, -5.484375, -0.4833984375], "student_probs": [0.44378021359443665, 0.02476281300187111, 0.0035535134375095367, 0.5279034376144409], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "41fd54780082bd534be8e680fac3389f00fed6fb347d3b1b2a92679d412ff4be:action", "state_id": "7322bdeeaebdca739710e947f459d8aa21984da360a29fefd3da732bb8de447b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.10546875, -2.12109375, -4.2265625, 0.04296875], "student_probs": [0.4853302836418152, 0.052366506308317184, 0.0063776420429348946, 0.4559256136417389], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b60ff8b00a8fbdd1728bb2421ec36a50c552231ce1361095ab487436476429fd:action", "state_id": "48563e5513421536ad49c5079b22867432b9119aa7e2a4bcd545fb70f2ca1465", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.7734375, -6.3984375, -7.984375, -2.48828125], "student_probs": [0.994716465473175, 0.00010337239655200392, 2.1166093574720435e-05, 0.005158980377018452], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e31af7519e868e957566cfe39fcd35639df60036d1a8b97de0e9b2dcbdd9929d:action", "state_id": "5ab27abdc325d3540de87f35b10e439afd1a42fd9e69c1d353048cd0c779780d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.203125, -5.23828125, -7.6328125, -1.99609375], "student_probs": [0.9845937490463257, 0.0005774247110821307, 5.267004598863423e-05, 0.014776090160012245], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8759eb489f6820b142b440c0f3ec42323c60e6ddd4d68f8b66b206474f2971d6:action", "state_id": "f31c3a74a2f686988c54f60f2ed6d64406feb30360eea1ff4621c37b27765ce1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.22265625, -3.5625, -6.328125, -1.484375], "student_probs": [0.9729362726211548, 0.002989667933434248, 0.00018815998919308186, 0.023885875940322876], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ceddb79b6e4598f62720e1cc4b01caef9fedb0b966a46e16c67a0f8141d1953d:action", "state_id": "efa50a406acbbffa0765bb08a2c11e3b877d9e181eb4d60e45808c5f64e167e1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.8046875, -2.294921875, -5.2421875, -0.583984375], "student_probs": [0.961378812789917, 0.005863572936505079, 0.0003077380242757499, 0.0324498787522316], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4c4d12b1ed0594e1f0e2afbccfabee381e0f128163ca54b3755f1b8a6a721059:action", "state_id": "444dc9d0432981b7e895842d0b4297ef170a26714590d39c2fb6ba65063febee", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.9140625, -2.115234375, -4.39453125, -0.5263671875], "student_probs": [0.9622193574905396, 0.006296195555478334, 0.0006444543250836432, 0.030840007588267326], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6102d67d726079de8e1221b3d22f812b53e608ddc1bb5ad47f9e24cf180858ac:action", "state_id": "d7742f28e8a584b157217e2b6f90f20278ab3e2edfb000161ce82f0e354f05c5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.83984375, -1.701171875, -3.7421875, 0.02734375], "student_probs": [0.9327468872070312, 0.009945481084287167, 0.0012918853899464011, 0.056015804409980774], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "05b622684f668b9a620015c6180ea82f3be7f5a86becfbcaccd9d1a954ff4411:action", "state_id": "d31cf31f0e9f7d652ada1ecb381eb932effbff2f61eb52b940c089b85b8b796f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.6484375, -1.87890625, -4.09375, 0.16015625], "student_probs": [0.7946908473968506, 0.023350290954113007, 0.0025491644628345966, 0.1794096827507019], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ba09d5ae63483f639be2e76c16e4191036b6b9b5c5077ba6d3029caf7d4712aa:action", "state_id": "eddc2b4450640ccfe0bc2ed607cb585ff9c80d3da398720230bbc9d07f9ca9f8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.41015625, -1.3203125, -3.39453125, 0.875], "student_probs": [0.35826462507247925, 0.06348496675491333, 0.007977175526320934, 0.5702732801437378], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8f92072b88e027bbee22f779acffa613400af0f95a5f2f849d37d5c442b2fb03:action", "state_id": "6ea7489e8002c4128bde766d3e45ef00c493576ec5b5aaf923c1b4d07fac9cf0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.04296875, -1.3671875, -3.0625, -0.166015625], "student_probs": [0.45474034547805786, 0.12096580117940903, 0.02220228686928749, 0.40209150314331055], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4fce44ec01dfea0f8bb4aca29df9e57403e70fafbe2e6379a8f22e12aa307495:action", "state_id": "c20ef5bf8ef6e59f6a13c789c7748fe497618f384fa9e0e0ea2426b6c5f3035d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.59375, -7.421875, 3.875, -3.41015625], "student_probs": [0.0015476824482902884, 1.2383796274662018e-05, 0.9977558255195618, 0.0006841023568995297], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "423e9adb0b2997c3b62e2aa556df86913ecabb6cd3b3cb33097c62c46a10ba4f:action", "state_id": "5d016a082bbdb829523b0fe2a684c4002dc749a51a37ea79784be96fdd0f0d09", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.42578125, -4.25, 2.96484375, -1.298828125], "student_probs": [0.012064829468727112, 0.0007161080720834434, 0.9735211730003357, 0.013697970658540726], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "86cb3da9745f358cf9717cfefe95039c01021110611a2784dcf652ba9412b65d:action", "state_id": "487a2365645c647845ae02cfb0398ded98af4c1c94fd5981a2cbae47b8b27e7d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.43994140625, -2.6328125, 1.72265625, -0.47607421875], "student_probs": [0.09285224229097366, 0.010361927561461926, 0.8072287440299988, 0.08955711871385574], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c03e77d00e365f0f292aa96ebc80cb1af5f4d0846296f2ee8f47671ddad4e7a2:action", "state_id": "df5b51ed9f1d4e9fd21f502e8713e4cee542844d3aa9ceb1d9b73a2aea1a0acf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.6640625, -1.158203125, 2.279296875, 0.91796875], "student_probs": [0.13369382917881012, 0.021612823009490967, 0.6723551750183105, 0.17233815789222717], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ea7d8b515e3b1eb6a0caf11c1207e0945cc0052e8ea725fee066c2e010a0bcb5:action", "state_id": "64f125ec3db15db02ab8a6a462d4e2e55d1616411a9385cea4189bde54d165b2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.2578125, -1.14453125, -3.0703125, 0.06640625], "student_probs": [0.4744560122489929, 0.11672551184892654, 0.017014125362038612, 0.39180436730384827], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e703826cb0cc7f3eecb77fd56c3385538ad39ff943274889d24a6aa1fb45beb4:action", "state_id": "c7a120c5511d70f896e14882c3388f6a9f006baa3e7069d76aee188b39055588", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5234375, -6.3828125, -7.8203125, -0.546875], "student_probs": [0.27285799384117126, 0.002116103656589985, 0.0005026186699979007, 0.7245233654975891], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2c8c07d632548a7bd1543e6f42363b5f19491f30326e768bbc49f69e4e38a511:action", "state_id": "30728eda700ad9b000c79eb4c0085eaa30ae638e8333a2eb4c949cbde46ceb1d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.705078125, -3.37109375, -5.6953125, 0.50390625], "student_probs": [0.2259165346622467, 0.015707682818174362, 0.0015371517511084676, 0.7568385601043701], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7b1fcf103c1aa53922a0e406666c25753594bb6f0b43a6078b292d94eb88a34f:action", "state_id": "e892618ae8f97a626b18b6d2fb9d2d134cefe5f5bf6f9e041ce316eeeb9de156", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.78515625, -6.5, -8.015625, -2.55078125], "student_probs": [0.9950957298278809, 9.233635501004755e-05, 2.0283607227611355e-05, 0.004791777580976486], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "15638056b9120b3d5c15c8d8e7757e9954c947d7a0dbf8e4ad9f2b2e5f838d96:action", "state_id": "9efe780edb5f21776454d1db8ad8481c2cc5c63fba16b54360e13ce60c9af927", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.234375, -5.625, -7.7109375, -2.28125], "student_probs": [0.9887570738792419, 0.0003817740362137556, 4.7412762796739116e-05, 0.01081380620598793], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "65c46289ae25f3cb652e7fe22af095c5efb27330c568a9eef658c994a2359f81:action", "state_id": "05abb2d3f4f709047f18de989743bcc78911aac549dac852c0bf790bdd68f888", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.21875, -3.95703125, -6.4453125, -1.8828125], "student_probs": [0.9815481901168823, 0.0020408162381500006, 0.00016949506243690848, 0.01624148152768612], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f2c10466ce6f4a5826a03e6459d86e4cff87599a84fca61b5a0f9750c68ca2ef:action", "state_id": "3c8a1e6ff79e645ff6e513cc31e6a493986a960409f0513df9e1251836823304", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.8671875, -2.46875, -5.1484375, -0.380859375], "student_probs": [0.9578584432601929, 0.004612465389072895, 0.0003163440269418061, 0.037212811410427094], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "39967bfbf62be32c4893e26352a9f2385150c5934205ee0260e5cc51a82da611:action", "state_id": "ff7502bfd511f2277d3f7cbff27c34bbb502c01e82a5ffda49ee180d4545857d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [3.0859375, -2.41015625, -3.6640625, -0.318359375], "student_probs": [0.9629238247871399, 0.0039506517350673676, 0.0011274678399786353, 0.031998131424188614], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "eb0c84c488dab99edc4a52b5252652270c30cbd58a21f838ef19bbb5f3403c2d:action", "state_id": "92d1bc3f1ad27136ac39225d3375c03d98aec96eaedd562b04e2b5ef6f59d7ef", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.95703125, -1.6328125, -3.58984375, 0.8984375], "student_probs": [0.7254548072814941, 0.02002446912229061, 0.002829001285135746, 0.2516917288303375], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5b9d74a62cf928e6e2ce81f4b93405a8aef98438904da411751edb4feb86490c:action", "state_id": "44922748e102c0bc77b6c257de4341d64c203bda345bffab1cf6d99ae6e9d589", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.12109375, -2.2421875, -3.4921875, -0.5733642578125], "student_probs": [0.6171242594718933, 0.05807812884449959, 0.01663966290652752, 0.30815792083740234], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "daa2958ee08891078a7dff2fe41862772ffaaea604be6b2f9a24aead605da5e1:action", "state_id": "41bd9de6bb45157015a6c08baba7d966a749ebeae8861220be7fb1855e6e9486", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.068359375, -5.7421875, -7.46875, -0.28125], "student_probs": [0.3117174506187439, 0.002910337410867214, 0.0005177340935915709, 0.6848545670509338], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2aba63c7851c4c770a5096ffe9daaae982a1b602f25d4a9e9748a5e655acaa35:action", "state_id": "6f4990036de1afab2156f2ec492e94764ea123863e868efe4957c50c64ed28a3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.29296875, -2.875, -5.2421875, 0.47265625], "student_probs": [0.30930572748184204, 0.023389775305986404, 0.0021926513873040676, 0.6651118397712708], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dfa2739eb3ee2ac606546f0e63e47e7beae60cc33d30e2237d922933ffed2463:action", "state_id": "4e69beeee2bfa9c2c08e4b6b5cd388025b2e283b46cafca51d5c700834eb5350", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.7890625, -7.7578125, 3.8046875, -3.57421875], "student_probs": [0.0013661609264090657, 9.497322025708854e-06, 0.9980012774467468, 0.0006230355356819928], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "be3f35e0b30839b3e46428c6d7ab0fdf0904acab7f040215f968b8c0491deab1:action", "state_id": "0c473ecea5b207fbe801781cae9203f2f28beb2002be44b58fae74018c537833", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.849609375, -6.796875, 3.2265625, -2.623046875], "student_probs": [0.0022850199602544308, 4.411784539115615e-05, 0.9948047995567322, 0.0028660567477345467], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e2b8377c122ee39e7df8850864cd76b732c7e8c09ad9bd3af21d5a15cd94acf7:action", "state_id": "0da2ca5b323af025aeba3aff0c8b93c2921da431f7e2971a1ac6ff448e571861", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.578125, -4.66796875, 3.27734375, -1.3046875], "student_probs": [0.0076452093198895454, 0.00034792631049640477, 0.9819574356079102, 0.01004943810403347], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "662717de0e2a1ad43b51edeedeb73f0e2e520a94f1bdc98d93c20af83c1ec132:action", "state_id": "07c2ad8f5dc68bb73e30127661b93512070d0317440c761b6ced2717459f8cfd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8671875, -4.125, 3.26953125, -1.71875], "student_probs": [0.005799754057079554, 0.0006065324414521456, 0.9868659377098083, 0.006727831903845072], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2b7250470292acb6a5d1c180dfdaa210d31d0581e51215e6b63b5fa93abfebfa:action", "state_id": "20202db43ff4eed7dbc88b40c0a2c3258c0ca148b080653655b71975f04ac919", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5555419921875, -2.34765625, 3.69140625, 0.015625], "student_probs": [0.013730843551456928, 0.0022876623552292585, 0.959673285484314, 0.02430814877152443], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cc5c1825dbd0a249eb3f8410252737e8cfea3e429ebf9c6f141247170fbf03d0:action", "state_id": "4006341fb21c6cd2f9d1381ba051dc097a3ec7cb88694211cd23b1809f1c997e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.53369140625, -2.53125, 3.74609375, -0.259765625], "student_probs": [0.01339123584330082, 0.0018167367670685053, 0.9671809673309326, 0.01761104352772236], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dae85ea13e3f82b84e8b3c5aa9ce5e36934f4e7c6483d5a98b72ab337bf0af79:action", "state_id": "9aa511814531c89080ee2826c517302c7f671e3fa3632b7cfa8326d6d3dd737d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.65234375, -0.3798828125, -2.90234375, 1.390625], "student_probs": [0.28759270906448364, 0.10244424641132355, 0.008222364820539951, 0.6017406582832336], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "861c00bc40d676bc2d2a45ed5b5c658b53426d78f43d385bdec6598a55aa7015:action", "state_id": "b4bc1046919259913b9d8fbb374ec9ed898ef0485c331d65ee8e66f6327bef1f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.34765625, -0.587158203125, -2.96484375, 0.390625], "student_probs": [0.40436896681785583, 0.15877899527549744, 0.014729139395058155, 0.42212292551994324], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0a64aa7e273ed6a7104d4275e974018bad372a3dd36ce1e03f7f45a7239fbfe6:action", "state_id": "4f3c64cd0f120ffc792672bc23dadbdc1a8cbae7762f98da9947cc23e3e27140", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.53515625, -6.3359375, -7.7578125, -2.0078125], "student_probs": [0.9892996549606323, 0.0001388867385685444, 3.350798579049297e-05, 0.010527895763516426], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1bd383a52a73f4b3e83a81ee422f9554265b589f69058375aed6335957b9f3c4:action", "state_id": "5d9f85dc16b0bd78833581992d86d69ce733776ad2b1dace391d6f21d628f398", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.25390625, -3.078125, -5.59375, -0.87890625], "student_probs": [0.9534524083137512, 0.004609217867255211, 0.0003724819398485124, 0.04156577214598656], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1454cb0c28ec48f986569bedece8bf88548be2031ec2c058ac48e94727fe3c26:action", "state_id": "2d40aa1da01fd00d492855b87462aea16c74c09630392c196f1bae529e32b97c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.140625, -3.57421875, -6.3046875, -1.455078125], "student_probs": [0.9699763655662537, 0.003197687678039074, 0.00020845317339990288, 0.026617491617798805], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "06a5a4eb0d9b9f000ea8f95a6e2769d960bbd676ce4f746fecbca2a59ed97d5c:action", "state_id": "b29097e377cf4403a3d08f8ab7c220c39adc594d7333617b6baa92318d5968a1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.7890625, -2.166015625, -5.171875, -0.201171875], "student_probs": [0.9454726576805115, 0.006663246545940638, 0.00032980539253912866, 0.04753425717353821], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c5a8ad33856b2e6bdf65dd65c6ec456723f0ebc0e95852781c62fb0af6c04cdf:action", "state_id": "a27c3c828fd5eb00474af8f65430b48e8c579c96132730eb40a5843b6eb42b50", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.9296875, -2.0390625, -4.3984375, -0.4775390625], "student_probs": [0.9608533382415771, 0.00667969137430191, 0.0006310922908596694, 0.03183592110872269], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dc79b45ff4aa8a88a5104ba6da4950818f7f82e42bb63b5ffe4381eecbd79750:action", "state_id": "7ed2f2693d57972ea6f572dafbaa0ccbf460d94025bd2ef74e8f39d6edbcce91", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.80859375, -1.6953125, -3.76171875, 0.13671875], "student_probs": [0.9245651364326477, 0.010230948217213154, 0.0012956480495631695, 0.06390825659036636], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cc634cf3fe799ec877c299423254ca76832081a99b50ffb13af084ce366e3fc1:action", "state_id": "91a17bb32fed8f98ed3a8be33043b410a711e312e0365a7fe584b094fb8ee634", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.69140625, -2.359375, -4.5390625, -0.4560546875], "student_probs": [0.9522997140884399, 0.006098839920014143, 0.0006896376726217568, 0.04091181233525276], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7eb065868aee761d29175813449d394f48acc47b574897df4eaaad6a3f16c4bf:action", "state_id": "c7cce27ec3467525207010cc037223724dcabf6eaf06faee926a8158cd3272ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.35546875, -1.515625, -3.51953125, 0.79296875], "student_probs": [0.36716920137405396, 0.056527603417634964, 0.007620353251695633, 0.5686827898025513], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6b3e68133b52f80330644c2c40b0b8cce784ae9fb62d17d8511f5ec2e024c0ca:action", "state_id": "9fc0d70c69e3ed644fee4e4cebd556d8fa6d1bb7a3d91309d1fb4cd8ec7de099", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.15234375, -1.44140625, -3.4375, -0.177734375], "student_probs": [0.4370834529399872, 0.1204291507601738, 0.016362102702260017, 0.42612534761428833], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ed9dff1fddbbe499bd2ed28bc52a457a6ff6fa571f7172e4d7c337d75d51c163:action", "state_id": "c193c304487c30459fff62d988ea0e96a31c091b98db21e08749117ed62ae821", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.853515625, -7.78515625, 3.70703125, -3.57421875], "student_probs": [0.0014121269341558218, 1.0188011401623953e-05, 0.9978908896446228, 0.000686872866936028], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4ea6f5ad0e24bafa81622ee426d41671a41c8010cc23307c52a96cfb4d954bd1:action", "state_id": "67f47e99c33053abf7a49239ff68490bb864a8014f6c6d7f3c1b3dcb5e3ac60e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.912109375, -4.984375, 3.2109375, -1.6015625], "student_probs": [0.0058734905906021595, 0.00027203719946555793, 0.9858419299125671, 0.008012445643544197], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "61b50f8f4fda40bf50ce978f7e9b2bd2d4361bae09545f4269512e7073b15d76:action", "state_id": "17b395d1b33b0364e98711fef350e3bc509f2e1debbc93d154dd4ae0008e0e0a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.72265625, -5.59375, 3.31640625, -2.431640625], "student_probs": [0.0023702639155089855, 0.00013424450298771262, 0.9943246245384216, 0.0031709042377769947], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "708ef8aff25b7eb782008553fdbe6405b9445bbb5ddfbbd3a5f3d724ba9f421e:action", "state_id": "8da134fcd4059d2347c136a7d60a6819fe6a59e302fb984715db530e660ebba5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5260009765625, -2.51953125, 3.55078125, -0.15625], "student_probs": [0.016249844804406166, 0.0022134515456855297, 0.9580170512199402, 0.02351960353553295], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c100318a36e3a245e196f2ab0930b9a5c2da609793073159b86287b829a4587e:action", "state_id": "60591ad4382476b8b3168dc7ec825f74ef8b18c7cbb842d2c47ca05628a7671f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.1015625, -1.056640625, 2.3046875, 1.923828125], "student_probs": [0.14877209067344666, 0.017188016325235367, 0.49548670649528503, 0.33855316042900085], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "19fa683292905c991721f404d33f2f7c59b0a84365d889906d4d64150753472a:action", "state_id": "a175b1c0cf3115c1dbbd7e990574a612ee5444357cd2cbb988ada4b58d824915", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.29296875, -1.18359375, -3.13671875, 0.0078125], "student_probs": [0.4968412518501282, 0.11348924040794373, 0.016096197068691254, 0.37357330322265625], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "888b533734e6964d797fcd2fcbaa5b4b07236c7cf91b549439d8088d99a05b30:action", "state_id": "adcd8f9b8f6a6499f155405b41b66f3442f196403757744d61bebb165babce4e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.994140625, -5.75, -7.4296875, -0.294921875], "student_probs": [0.3308645486831665, 0.0028458156157284975, 0.0005305517115630209, 0.6657590270042419], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4247ba80a06377efb75cd7391e14d10de334c3d0d0d9af99b5ad59c544c90456:action", "state_id": "bb5952be7b150013cd3469376088f545ebe755a218e6cf45156a4a2802945f7d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.220703125, -2.93359375, -5.28125, 0.359375], "student_probs": [0.34978917241096497, 0.023206677287817, 0.0022183945402503014, 0.6247857213020325], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f8ac5562f53e7ca9db9e78e2f5864c91ab4c3f234ba49433f7697035af3f6c1b:action", "state_id": "04b09c0ed6902d3bcff39fec7d624552edb4cbfb7bc81938e555484703737944", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.421875, -6.125, -7.625, -0.5869140625], "student_probs": [0.30158504843711853, 0.002734441077336669, 0.000610136310569942, 0.6950703263282776], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "21ca18c5b3554869c944eae3a0ab655ab88cb0c4ba9de43d98c0466d7e4bbe56:action", "state_id": "b65a00bf0ad15185986309272e688168328a697143f3df25d7f38b34a26c71d1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.576171875, -3.16796875, -5.5, 0.453125], "student_probs": [0.2576487064361572, 0.019294116646051407, 0.001873426022939384, 0.721183717250824], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "101f8e754c3dea5b65fe76d2c2acfbeac05cc4d1b7a537aa235a5035e83ec016:action", "state_id": "26bb2339b2c286176dc535a8780d854a3accc6adf77a74a0f00fa3c0bc74e7f5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.53515625, -6.59375, -7.8203125, -1.7421875], "student_probs": [0.9861739873886108, 0.00010698427649913356, 3.137838939437643e-05, 0.013687582686543465], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d0f6373bae2c3407f38bbf6ad9b7c4a82da822ba9a270f9459a0b821ac70dd27:action", "state_id": "eb9a54b13b619493f7a3252d7fbdb72bdaabc46bb6433878e9ca01653f71b78f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [2.01171875, -5.703125, -7.75, -1.66015625], "student_probs": [0.9747229218482971, 0.00043487767106853426, 5.615915870293975e-05, 0.024785980582237244], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b264ae8d48b897ffbe38c117ae8c5a484169884de525312be456721ddd1e7f8c:action", "state_id": "8327668d94e97e452c8107df0876ef828be9208b1279adddce5bb10dab130973", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.58984375, -3.109375, -5.0625, -0.380859375], "student_probs": [0.8697525858879089, 0.007916823029518127, 0.0011228444054722786, 0.12120770663022995], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "32f3aeb4f85e6d085ce7e50673fea3ea259925e9e1fa03f6b6870b4f45de8132:action", "state_id": "25bf6c0f1af769dc81cd8e0326d5e4cb46a1ab460daf8711489008c7343f2522", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.70703125, -2.84375, -3.234375, -0.208984375], "student_probs": [0.690496563911438, 0.019818775355815887, 0.0134100541472435, 0.2762746214866638], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b643ef924c204d8d985159035095a637fd3bc515d38a1cec0f702dc5ed38b8e7:action", "state_id": "7904f49573877088ac21a6a2c7238c285c206e5a8fba034fec8c607a38240408", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.04296875, -1.63671875, -3.3359375, 0.89453125], "student_probs": [0.28059759736061096, 0.05231243371963501, 0.009564089588820934, 0.657525897026062], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bc99e47de217fe40249bd59a09652808b3224ea45acf69a25e7f9383c0af07b0:action", "state_id": "a5f503c003da54b3848e2e6694da16ad15408a49a69caa9ac9f0b3852f1b7697", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.2109375, -1.1953125, -2.9140625, 0.6640625], "student_probs": [0.34937936067581177, 0.08561909943819046, 0.015350657515227795, 0.5496509075164795], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6a3ffaebde1fd0e46b9fab85ea33f60ab2271faeca39c4c3dd1c6a2c4bb463aa:action", "state_id": "92f3227ef44c1f088a18ca156e6d8684b35c3d3474bd32d2a28ab355d48c5353", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.47265625, 2.57421875, -1.80859375, -1.9765625], "student_probs": [0.006245291791856289, 0.9713655710220337, 0.01213253103196621, 0.010256602428853512], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a43e7d646c8b3338e35a79d3058010f2818979e23e827b19052d8c5533dc3432:action", "state_id": "0e954dd1656b3cad1585b530d159c3c8004f0f6d586c15fa5006b9971bacc028", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.84765625, 0.91015625, -1.78515625, -1.134765625], "student_probs": [0.050327908247709274, 0.7934355139732361, 0.05357378348708153, 0.10266286134719849], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "788771109ff108a2c47d0b9baf568ba139c5a1f1bf18a52121ec3014c1136047:action", "state_id": "584fb2941abeb171dd6c7edcb6a8f1165da11944e06c3cc0d2150f7e7653cf44", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.01171875, 1.171875, -1.8125, -1.048828125], "student_probs": [0.03451485559344292, 0.8329582810401917, 0.042123615741729736, 0.09040327370166779], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0167d10f053233b0beea21a5172e42d2e94a0e5eac9c2b18b6a9ba780fb83bac:action", "state_id": "37eb352fc6f1938dd09dc048590d6adc3f191fcc64abb9432f73e3ade4dcd8eb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.58203125, 1.220703125, -0.1953125, 0.56640625], "student_probs": [0.08553137630224228, 0.5188514590263367, 0.12591436505317688, 0.26970282196998596], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8306c7c86bdd93644074d264b20131e0a5bdface9071137cabb108329ecf9a92:action", "state_id": "c393d67be01fcf09b810d567ac68788a127ff0bcacfbd73732d26dee90b4f9bc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.31640625, 2.80078125, -1.3046875, -0.80078125], "student_probs": [0.015367398038506508, 0.9433485269546509, 0.015548545867204666, 0.02573554962873459], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5ed94ee3f8146f18b36f77554949c1ba015ccde41bd342a1f080db9d5bad8d66:action", "state_id": "eb8cadcff8cbbd24d49da5e028d3118d81b7e83e2a9d3a611a41363a755d8118", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55859375, 2.689453125, -2.1328125, -0.939453125], "student_probs": [0.013626002706587315, 0.9533926248550415, 0.007673410698771477, 0.025307999923825264], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a163e18c43a056e27d3482c9cd839781772a7207a1b84321671fd490c778a166:action", "state_id": "407c0a9808f9323a4bfed7d5e83ccc40067293c247a880569f6db05bc981087d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8125, 2.66015625, -2.3203125, -1.53515625], "student_probs": [0.011048441752791405, 0.9677227735519409, 0.0066490694880485535, 0.014579743146896362], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "51adbb7f2685badfd334acb97934b66e7bdb7e7ba2159309d4a2f8750a326566:action", "state_id": "8ff97a523284a5e6713c38b2b3a537672db9474a447b03660242ba85b8412627", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55859375, 2.958984375, -2.20703125, -1.26171875], "student_probs": [0.0105840303003788, 0.9696395993232727, 0.0055339885875582695, 0.014242369681596756], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c91c7689012f458fe989eba59dbf9798b510c081f129784b300799d079689955:action", "state_id": "78f0c5065c443fad50cefbac4102400036f5d6261d78d0a4eb21b805cbfdba82", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.142578125, 3.107421875, -1.65625, -0.978515625], "student_probs": [0.013720810413360596, 0.9619030952453613, 0.008209087885916233, 0.016167066991329193], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e222e70bd851a9eba46995cc1acfde8f345fc57d969c7a0b9ae0fe1c7c397e60:action", "state_id": "c83674997974b670668ae9d9fc1883e4bf15aea0c160ce8a8ba7501761b28416", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.17578125, 2.9765625, -2.15234375, -1.6328125], "student_probs": [0.005663126241415739, 0.9787926077842712, 0.005797423422336578, 0.009746856056153774], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "05310651c1f97a9a35501f90fe9420e7d4c29ee97a0fbed76df3a9e4f515d960:action", "state_id": "a64896b1366ee6eb7567ccd353dbbf4866210de6dbdf2d71b21b509bba426692", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0078125, 2.857421875, -2.09765625, -1.6484375], "student_probs": [0.007516093086451292, 0.9748473167419434, 0.006870265584439039, 0.010766306892037392], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cec68a72e5d1366f1dbef8fe3a70699ee0b8ee48b1677c06c0868ba53cf7394b:action", "state_id": "bab5d298cd8e026388e3a765200dd818f08d6fd713c18b91f0e27527b07c56d3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.625, 2.77734375, -2.51953125, -1.19921875], "student_probs": [0.011822903528809547, 0.9652454853057861, 0.004833193961530924, 0.018098335713148117], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "992584d5a5b703fb7fef232e63d4746c73e94f1533cccc4174b7ae06c12330e7:action", "state_id": "17ee92c71ce4f2a6e7bc9a6628d72a78d28439e551cb23b10ecde4629da1066e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.64453125, 2.8359375, -2.4453125, -1.017578125], "student_probs": [0.01091739721596241, 0.9637446403503418, 0.0049016717821359634, 0.020436258986592293], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "da45a5f5ce5f1efd3548c6e9d4d05610f23575536d571f44e350f459f6329360:action", "state_id": "44f367a7a6b10b15c1b9ad7338f266d9d33f5495193dea6ac6de335593e76f58", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.98046875, 3.09375, -1.91015625, -1.037109375], "student_probs": [0.016354799270629883, 0.9617361426353455, 0.006454863585531712, 0.01545419916510582], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4ee4d5eaf6c7cbbada5c81f1c93868cff13b52151a07a1a7247848188bd2e445:action", "state_id": "6a64b24c5591bd54e47008ef4531d54579fcd4d7e9444c1ab0c9d78c0c141bff", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.234375, 3.337890625, -1.5859375, -1.0390625], "student_probs": [0.010031864047050476, 0.9707141518592834, 0.007058297749608755, 0.012195644900202751], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e898d63e44f8aa637461524dda243a1e8322e5236473ec65f36215411fb5a678:action", "state_id": "5d98be076c60d31f22defea53d5f858c8fcfdfbc3ed004a62d1dcc542e0d8539", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1875, 3.1875, -1.58984375, -1.111328125], "student_probs": [0.012167264707386494, 0.9665655493736267, 0.008136868476867676, 0.013130279257893562], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "57b832e4f9f4247767c3683bff31c071ae0c23c9bc3499214afbb63a15d1634b:action", "state_id": "9e476843c4b838abd630acc1f134cc9e54130bc2afbef94f2c0cdad35ef7a285", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.140625, 2.998046875, -1.8828125, -1.35546875], "student_probs": [0.005715068429708481, 0.9743573665618896, 0.007395848166197538, 0.012531713582575321], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "346312faeeb8509c4fce2f08a5a0bded9d50db477ecb79a515390b23e8cceea8:action", "state_id": "da7fb1e078f12316c55a55a6998f97e05f7630c6ef21c9dcc89d8adbd472ed23", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.69140625, 3.009765625, -1.640625, -1.1171875], "student_probs": [0.008779329247772694, 0.9663941860198975, 0.009236667305231094, 0.015589828602969646], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "70f0eb8107dec99f16fbf459f4cbdba5fde88623c9bf2960b5fb827829035e58:action", "state_id": "a0d948e37101f68635f12d08982661dbfe6df5a4c09eb1aa12092dee0f67e5a4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6328125, 3.080078125, -1.859375, -1.033203125], "student_probs": [0.008696232922375202, 0.9685310125350952, 0.006933241151273251, 0.015839379280805588], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "31accba9b58f61544d965e5ff606a2b0677fca6875fc0f6e823aaa242632818c:action", "state_id": "dfed33f08ee732b28cbb533c92d95f4002992556ea4557312d9d842712f4aee6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.962890625, 2.9375, -1.70703125, -0.20703125], "student_probs": [0.01885855570435524, 0.9320228099822998, 0.008960500359535217, 0.04015817493200302], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fdeed4c7e2c7b24f62d64ea078409e2c095bc0682c23a7f2c9c90d12b31cef2f:action", "state_id": "47c3df7728f2f2f47cb74ee33ce2d00a398f15e9e8db63efa88da7251d60ae24", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.93359375, 3.001953125, -1.66796875, -0.5615234375], "student_probs": [0.018477225676178932, 0.945851743221283, 0.008865470066666603, 0.026805559173226357], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fd138b77e9a0c4bf817c46a1c45b2c4dab47a33aec8c675e3a86e437d4a4168b:action", "state_id": "e6e2f0bff6b0dfb0cdf4aebd2915669b3dfde89cf827f97179f990c5eb9e0e08", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.04296875, 2.828125, -1.216796875, 0.48828125], "student_probs": [0.048387348651885986, 0.854342520236969, 0.014960454776883125, 0.08230965584516525], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "430e8913ee6810fff90cb2c7e6480cd88b61f807a0e6f510a250f12cb8f3a5e3:action", "state_id": "a580c4a7c542a0e4c4fc5b06aa93122b93000fad2dbce3eb84e2f63fae4a3d84", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.44140625, 3.06640625, -1.15625, 0.806640625], "student_probs": [0.060798417776823044, 0.8392962217330933, 0.01230379194021225, 0.08760149776935577], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "411ba42154e6651e15112391fdfcb4af1e9b32afcf33738784b1fff43edf2be6:action", "state_id": "c2b074391b3b446116b874bb4d69c24feb74a926c1b9ba4d5a7469ad1764cf15", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.55078125, 2.8994140625, -0.58203125, 0.9375], "student_probs": [0.07538343966007233, 0.789358377456665, 0.024282965809106827, 0.11097516119480133], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "641194c7f0fffc874cd46cfd3abec21eda2fe86073938e5cc49a03492b5d7bd1:action", "state_id": "bf61bb414ba5ae96abb84a76045da0325974ca5b5026990758ae469a8fe4f896", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.5625, 2.537109375, -0.6552734375, 0.86328125], "student_probs": [0.10151658207178116, 0.731305718421936, 0.030037563294172287, 0.137140154838562], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1d3aad0a62a7d045633914474544a7e84dd327e3c3c32827c17396ff413d734c:action", "state_id": "5572c07e2208c999136d55f1e7ca1bcb239aaa8bbbf7904e22eedfa7abd510b3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.28515625, 2.81640625, -0.80712890625, 0.80859375], "student_probs": [0.06413348764181137, 0.8061071634292603, 0.021513517946004868, 0.10824576020240784], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "762450c7e45461ed6c2852402bb0fc38949169bd2dd636dcf0ce2a6d12979331:action", "state_id": "3884f765a4e2b3d820a65857b3f41c171536c38da50406676273b14896077f61", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.07421875, 3.072265625, -0.990234375, 0.814453125], "student_probs": [0.042575493454933167, 0.8534830808639526, 0.014684987254440784, 0.0892564058303833], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "305b33bd08fd63900afae99790d45b8374b2e93b1a9e81df1b796e4319f2172b:action", "state_id": "05e7a64e84d7c82a85e61a0e461f4aca333bae484c017d230be87c418fb8a588", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9765625, -4.1953125, -1.55859375, -0.953125], "student_probs": [0.38131821155548096, 0.01525464840233326, 0.2130662202835083, 0.39036092162132263], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ee1c07c0f69b7a7bb5e7f5b69e07eb941c2d48234b6a177311675490b04e34ed:action", "state_id": "7deaf85a27c183d17cd7dfd4f3c9138e9a2e52d0174fea2cd9fd94aa52de5221", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3583984375, -3.46875, -0.578369140625, -0.6220703125], "student_probs": [0.38235753774642944, 0.017047517001628876, 0.3068581223487854, 0.2937368154525757], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1e2289a2e20ff492cc65e4cee6f66385895c7e40bd4be9773b828dfb68d53ebb:action", "state_id": "414befd3a03644753ab4098d39218108912dccb0efa25338fdbd55a4e1f37b6c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.08203125, 2.63232421875, 1.36328125, 0.3671875], "student_probs": [0.05335620045661926, 0.6835386157035828, 0.192143052816391, 0.07096213102340698], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f99d9b6d4ad4c73bb3ee89c388f1d76aa38451f2d11b5b3bfefcb7d6e7f39f6e:action", "state_id": "2ea4fed87e0559b95f5f5bfe82a3cc66f48c113085ccc8e7343f9a3251562235", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.26171875, 0.642578125, 1.06640625, 0.4375], "student_probs": [0.1697298139333725, 0.24840667843818665, 0.37951546907424927, 0.2023480236530304], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "11b4c8dba17161e233707607b29f0f430021f131b237237fde67382e58a37335:action", "state_id": "7e4a28b341800de5cdeb278114a26d048684706f2b9c41f734aa4e8709995590", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.2578125, 1.234375, 1.181640625, 0.5], "student_probs": [0.13425953686237335, 0.3565010726451874, 0.3381882905960083, 0.17105108499526978], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "da1b8506d5342633083844be4a6e5ea9a8484c859e384ca231f84f526c301baf:action", "state_id": "020e4972eea847fcfc6f00ce7af5a2055c7b992bd8a97d510ad098c5d52b91a0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.46484375, 1.92578125, 1.087890625, 0.63671875], "student_probs": [0.11958678066730499, 0.5154188275337219, 0.22298158705234528, 0.142012819647789], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0e4d56b193dbac76636df0696ccb9cea9bf6c29c9bc08bf93eaf8d97fb9c37fe:action", "state_id": "3a98927f6ad4860ad41bcca5bbe70d7df3c56c7c6b1f5dff361213ddc9d5170c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.46484375, 2.9654541015625, 0.83984375, 0.728515625], "student_probs": [0.06270919740200043, 0.7644208073616028, 0.09124134480953217, 0.08162862062454224], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c147eb3c3a8a4b4347c575fef997d200a1fa8f791e9fdf7aa08b242c24d60793:action", "state_id": "de930f683429bad977f29b9c4b717832239aa278fe078b1d26bb80e46d4a71f9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.2734375, 3.199951171875, 0.720703125, 0.646484375], "student_probs": [0.04409436509013176, 0.8229090571403503, 0.0689648911356926, 0.06403174251317978], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "26f6e8acf124351b4d93b0e9d396cdb472bd4db0c9da7bbbf1b3755f467b39ad:action", "state_id": "36c0651accd543717a16cabee0f823d1f2a4d14d839522a857612642ac6dad3e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.896484375, -3.53125, -0.53759765625, -0.7783203125], "student_probs": [0.27556565403938293, 0.019767917692661285, 0.3945368826389313, 0.3101295232772827], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b7a00226575b536a81637e2f36159ed58626df2f3a1eaee35184660aa7b6ba83:action", "state_id": "e517b55a7c5700bfb70e7d47e95994360239ae8821fba83212e501c9c3216408", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7568359375, -2.78515625, -0.224609375, -0.56005859375], "student_probs": [0.2468070238828659, 0.032469019293785095, 0.4202430546283722, 0.3004808723926544], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e815c9c08500f0b54068c761d386da75ee467905d3f53976377e5dce95ccd9e5:action", "state_id": "85c7f14ad74867f678cb0dec1d8eae503b98d9328d0b70e290de613c090b2b95", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.6484375, 2.46484375, -1.060546875, -1.4296875], "student_probs": [0.005698290653526783, 0.9471405744552612, 0.0278841070830822, 0.019277069717645645], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2416c4f3d7fc9355637efee2dfdc62ec7f0ed6acafd5229784348f24029e829e:action", "state_id": "eb03a4996d2587c8e128918e5e069239843ce0e6b23a5f3d8ca612f37279b3c4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5859375, -0.13671875, -0.111328125, 0.20703125], "student_probs": [0.06395343691110611, 0.2724279463291168, 0.27943360805511475, 0.38418495655059814], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3ea899d382e56227d6798f7ba58d4c12760861440d2bc002fa14a2a0dc3df30b:action", "state_id": "ab87e78b06c2590ae66656286f8d33cc4ab7374367a7aa5bc2a30c448514dfcf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.265625, -3.42578125, -0.8125, -0.6494140625], "student_probs": [0.09412761777639389, 0.029503095895051956, 0.40253275632858276, 0.4738365411758423], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cfe1c419479cec10cee20d679b5df808568f61eb25216a381d69064d92430daf:action", "state_id": "8f2845c3a66e44314d8ec1b8d697bcc3e4db3cf1c0934d1818829674079e6ab5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6796875, -4.03125, -1.033203125, -0.8251953125], "student_probs": [0.18676860630512238, 0.017784155905246735, 0.3565073311328888, 0.4389399588108063], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dc5a89e0e0a3585e9e2a1c061cf68e1adb3038338bcf466427662226f3300020:action", "state_id": "c4280f8326789bc3acbd4a418fe9457e5de1d47c826e9bc4ab4f9117fab7c31f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.41796875, -0.296875, 1.716796875, 0.00390625], "student_probs": [0.17196230590343475, 0.0841357484459877, 0.6302417516708374, 0.11366014927625656], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "82078143802d070828712183dadd73e2099844ffb9e34f0f2e7092573417993b:action", "state_id": "7f15d0ad24146630bfa8a8a9d021b0cf5a0c72f8e1ede880fbf4e68d7e58d4b9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.8046875, 0.4296875, 1.40234375, 0.8125], "student_probs": [0.2215828150510788, 0.15229149162769318, 0.40280503034591675, 0.22332070767879486], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a304f09748a55ca4e7ef8f8f5c6134d6d3005ce9a313ff7304c7922eabed2c6e:action", "state_id": "75ffdd8f60b7a5ceee768e9ff18165d6d2ffb5fe9a647db09ab19751a6a27eaf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.30859375, -2.859375, 0.02734375, -0.203125], "student_probs": [0.278667688369751, 0.021741842851042747, 0.38992616534233093, 0.309664249420166], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "89ebba61d4410e84fefe831b9a441335f3f8eead1018bfe94e7847769c39ed28:action", "state_id": "3b8d69b6c0dcb256e7cfd9dfe50ab2f0657b708c9171f602eb39e9a56d97d4de", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.01171875, -2.23828125, 0.203125, -0.015625], "student_probs": [0.2990727126598358, 0.03226955607533455, 0.37075093388557434, 0.297906756401062], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "984cdb32aaf2ab8cb7a877cf662200b29cda37d2d2c60c4ebb2e34d51ac834cb:action", "state_id": "21f7612ca4c3b82e9446d1550f1747821b88895e8def9711d1c120c0efe2f0ee", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7734375, -2.62109375, 1.08203125, -1.28515625], "student_probs": [0.12267281860113144, 0.019333986565470695, 0.7844552397727966, 0.07353798300027847], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cb771b9043624448a8b8494a91509a6329544bf866eba4eb99923da65152890e:action", "state_id": "9a2ea53079f05aca177afd5d3c0db47f320fa6c20237f5a4b8b66a9bf24d3a50", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.4609375, -0.287109375, 1.3671875, 0.60546875], "student_probs": [0.19593198597431183, 0.09273266047239304, 0.48493632674217224, 0.22639897465705872], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3d6c5fda23b5e95c7244b990995f807b91c685239efc0fdeceecad2002875025:action", "state_id": "ea08f46dceaf95a2ee6652fc4aa0c4d44bf228eaacab3fbbce8201ddacc4adcd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.498046875, -0.2265625, 1.97265625, 1.314453125], "student_probs": [0.2764032185077667, 0.04926684871315956, 0.4442867338657379, 0.2300431877374649], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "94ebce94befbf2d54677db37fc06ca6bbf4659d60eca894a3a88112eae2187a0:action", "state_id": "f8b6cf0caf5bc47f669734e51e2de56da05a0a62ce56c871098694c7f5bdfb64", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.115234375, 0.29296875, 1.8779296875, 1.017578125], "student_probs": [0.2226952165365219, 0.09786005318164825, 0.4774690568447113, 0.2019757777452469], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e806266ae04272ba38d0330a26f11a17747ed4a30d553c4484b7b62cd6070097:action", "state_id": "9f30b4ad38403f636ae87448225a7a0fa7f71c256a6c59d23d822a5b925289ef", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.064453125, -2.48828125, 0.30078125, -0.046875], "student_probs": [0.2819150984287262, 0.024972565472126007, 0.40619784593582153, 0.2869144380092621], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "eaf7f6cdbe0cc1215c2a9029a77f1ea4f924107cc1b9786b5ed829392bf99aa8:action", "state_id": "a96d4defc16843e6cca154dde77e2ea37c6b91d82865c38280f3c15c93db895b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.283203125, -1.94921875, 0.0390625, -0.3662109375], "student_probs": [0.28656628727912903, 0.05416062846779823, 0.3955335021018982, 0.26373955607414246], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "797ad33f06856d0127b6faad517d5a06332ae49ba351f655c7bae86be6d583ef:action", "state_id": "d2d6b37ce87eebd9bcaf6a43c37d5514ac5a913bab37821563c12afef7311587", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3984375, 2.26953125, -1.10546875, -0.92578125], "student_probs": [0.0231927577406168, 0.9085126519203186, 0.031087592244148254, 0.0372069776058197], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "482c840fb548fe815c3404871b5075d2b980804bd2408eb59c82ee3243d01f83:action", "state_id": "62cb2cb6a8ce6d70fd530678b1207bf1aa26f4a1d365940a4233f6ca7c5a7ec6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.62109375, 0.89453125, -1.4375, -0.06640625], "student_probs": [0.12927111983299255, 0.5884764194488525, 0.05714007094502449, 0.22511228919029236], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4dfee1441a791830f54684e0f26e11e9f8df8d49246ed78bee2cabb9e9cdc755:action", "state_id": "ce62266ad331580f20d0bd4ab722b5216c1039a2e488acb23316550b71f10d41", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.275390625, 1.216796875, -0.904296875, -0.150390625], "student_probs": [0.05676262825727463, 0.6861289739608765, 0.0822671502828598, 0.17484119534492493], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "db71e09f4edff1b81627dd90173bc8c44ae67d5915348fd8a3902ba7b7621b4a:action", "state_id": "94c4ebc193cd81c577856724d665b37f0c3af1eace1b2a3f32d993674aa19646", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.63525390625, 1.875, -0.4296875, 0.23828125], "student_probs": [0.05906102433800697, 0.726926326751709, 0.07253996282815933, 0.14147265255451202], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "782c021e033d9e7bc984d43c5f9ff29e6baf0a2f31232d596b632678d667c22c:action", "state_id": "a16c27c3d2fdada0f985a6d89dd4a1a1e5dc9c431a1686f84fe1e73fa30d25c5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.98828125, 3.001953125, -1.62109375, -1.3828125], "student_probs": [0.006611716467887163, 0.9717296957969666, 0.009545126929879189, 0.012113397009670734], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "936c26ad028e36f41ad57e28378a547fae7245c10eef320478aae2297252a2bd:action", "state_id": "c3366767f21ce69f7909da9e57104a630ef3658e5bf32aa7db0c16a514db1936", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1796875, 2.91796875, -1.62109375, -1.33203125], "student_probs": [0.0059269689954817295, 0.9698768854141235, 0.01036159973591566, 0.01383455004543066], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "65ea654295d46c08de7a20af2dcb86dd219ff377cab946b6ac49ce40281ade83:action", "state_id": "32582a600b8594b201a215b8530dc1c63f6e3527fc3c0eba8a4aa8399976a6a0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.212890625, 2.724609375, -1.05078125, -0.39453125], "student_probs": [0.017942694947123528, 0.9202847480773926, 0.02110041119158268, 0.04067210480570793], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "01df648f4a6e2fc56d5b8ad61a1c181de10af3bd9e2f2a3a03f80074f764bedf:action", "state_id": "279e1e03c090a71a97274be8437def747973bc5eace4c923b7f9485cfe04c36a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.49609375, 2.822265625, -1.57421875, -1.078125], "student_probs": [0.012737376615405083, 0.9561359286308289, 0.011780147440731525, 0.019346460700035095], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "eba736dc362d1318183ce1bb7cf8026c6d1b8b6ed020a774e8ddaf20474dbe1f:action", "state_id": "f3854f7c53ba18a44f31850761dc15c659196fb3eaf0e60856fe0eb39c690517", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55859375, 2.73046875, -2.203125, -1.048828125], "student_probs": [0.013142693787813187, 0.9580773115158081, 0.006898711901158094, 0.021881284192204475], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7d15549d9b978ccaba571e376a959c0f55f8e72a4d16c605b7f0559288501b44:action", "state_id": "922705297078b416f5f8dc3e7514aa4ca6330797ccec1197a68987cadfacf46a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8671875, 2.802734375, -2.34765625, -1.5703125], "student_probs": [0.00911963265389204, 0.9729682207107544, 0.005640432704240084, 0.01227180752903223], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "91cae6a5d1c016a5a7de2be016dc5b9c1be30fc7b1cd638aceb609a1d07f0f84:action", "state_id": "213629748de3938546ed559d9345f0660b3508d6272a92e352d94d72441731f6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.875, 2.65625, -2.61328125, -0.6435546875], "student_probs": [0.027320247143507004, 0.9334412813186646, 0.004803509451448917, 0.03443499654531479], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6b37adfb3ac0658fe60c4302758ea1b075287c9153936ec76106ba6c7fbbb8a2:action", "state_id": "d1d1a6be8ce88bfe6adae3e58dc101e782a9aa71af85a7590d67dd73023c092b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.67529296875, 2.671875, -2.0859375, -0.47412109375], "student_probs": [0.03237403556704521, 0.920138955116272, 0.00789881031960249, 0.03958810120820999], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "785f8d3e8b01c4e7477a1eef43359ab804b09a9722e0b6f7715dc6ec47bec83f:action", "state_id": "e626c3b9076718ea8ad2d258036094b2920c713555a2699819416c23607560fa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6171875, 2.80859375, -2.3671875, -1.1171875], "student_probs": [0.011534134857356548, 0.9640008211135864, 0.005448339972645044, 0.019016573205590248], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ee1fc62ae8d40c094d6b3950e985904093d9b57916ba07a40e3209af87054b57:action", "state_id": "17964662d4d3bb25b5026c84f44c9c785fecc849c28f1fb3a1366dc7f12ba1e4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.01953125, 2.849609375, -2.28125, -1.5546875], "student_probs": [0.007486685644835234, 0.974833607673645, 0.005762707907706499, 0.011917047202587128], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b8ae3134dacf047a749dbc740b3c2d9924218f881e9be1a4ef44e8d19e991afe:action", "state_id": "dc7354597f3eef5134459da62f3113e03ace3242df3ea767055b67e34d92840c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.296875, 3.03125, -2.62109375, -1.81640625], "student_probs": [0.0047757504507899284, 0.9840494394302368, 0.0034533070866018534, 0.0077215866185724735], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7d9491b68d8288102d17121227220266c3a5055b9a604005dc4af5d56060fa7c:action", "state_id": "bffa7a53f4827d3176ebba446db581c0d743a89d4c73342712a965554bb17175", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.81640625, 3.0546875, -2.33984375, -1.25], "student_probs": [0.007472839672118425, 0.9749330878257751, 0.004427511245012283, 0.013166573829948902], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "596909f2cc04ad6d749c5b9bc8518715061b7fba3c6799a165e9b79010fb69e8:action", "state_id": "e6ba79686d62c24185b4f83c1ff044fc6352fc49dc24ca9c50d82f986a174166", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.947265625, 2.953125, -1.97265625, -0.4033203125], "student_probs": [0.01904645934700966, 0.9413093328475952, 0.006831133272498846, 0.03281305730342865], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8b9c184995c79fe6f3760427cc118ee03ccc693af1da5696909ecfe5b826f61a:action", "state_id": "73a3a68d69a82111a96c5ba3a7614253e1a896d0fdddd1863496f07414dea12c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.34375, 3.048828125, -1.92578125, -0.900390625], "student_probs": [0.011909686028957367, 0.9628812670707703, 0.006654682569205761, 0.018554480746388435], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bfe966d2db4c293c3b965390bb1e784b589d321b1290fcc56c008d8ef1d75255:action", "state_id": "86f6119395ac5b7dbe4d4547b44a0ad653a2de5fd5b6a3b009c8a904d09bb7ae", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.205078125, 2.998046875, -1.68359375, 0.078125], "student_probs": [0.03681252896785736, 0.9059311151504517, 0.008392367511987686, 0.04886402189731598], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c009fb6e37cc176988d2bf644ec2fd2017d2237b2d7d222b794fee72e92b42ec:action", "state_id": "9ffefe4625804e8af2b253f73f594d54474f37c6f3306deb4ce1cda0d40acfcd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.212890625, 2.9921875, -2.05078125, -0.208984375], "student_probs": [0.03728492185473442, 0.9193502068519592, 0.005933999549597502, 0.03743085265159607], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e7a4a6261427bce9eb4e09b1461c2c0c093fe477b87bf7cd8d09e36b45392f0b:action", "state_id": "ffa4e0054be3ebb39c943511b1872508aa4cb4e1f5d2b883710468c2888ba5a6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.240234375, 3.107421875, -2.01953125, -0.2734375], "student_probs": [0.032709553837776184, 0.9301291704177856, 0.005519958678632975, 0.03164132684469223], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "03c214e8eeaaa4bd22325174897f5def323d2afd8ce9c234711680934457eab1:action", "state_id": "838909492dc5ba6c6e86dd3f566e62f56760de72bd44c14ff2e2ae8b54a9ec42", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.29296875, 3.2109375, -1.953125, -0.201171875], "student_probs": [0.028144188225269318, 0.9356552958488464, 0.005350471008569002, 0.030850030481815338], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c6f8ebaedadb4a38cf3e20621833d929c1c8d46e99471dcb1f0097119a6741e9:action", "state_id": "7fe43f06d7d87afcfcaa2372b93a14b8c3d7b3167fb274cb2d6913e352556576", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4560546875, 3.2421875, -2.09765625, -0.15234375], "student_probs": [0.023296549916267395, 0.9406276941299438, 0.004511833656579256, 0.03156396374106407], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "79e06620d18ac57676ca35a753da9e77e2ee15014787cfd086735b9ca47c9ce6:action", "state_id": "e1df2078880a7c438bde77fb9a6c4f47bb31418980bfeb79dc19d5c2e2156f5c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.46435546875, 3.162109375, -1.953125, -0.1484375], "student_probs": [0.024889925494790077, 0.9353566765785217, 0.005616415292024612, 0.034136973321437836], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5d96ac9feb27d0765403a2e3be6ca1884933348d7bc7f38fcf26d5be839ce8fd:action", "state_id": "fea0493cbae0cee6c1f6206ec0063a4d296f7646e5335e4f6f0ceb45689a4f47", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.30859375, 2.66796875, -1.072265625, 0.853515625], "student_probs": [0.07374539226293564, 0.7805458307266235, 0.018536821007728577, 0.12717197835445404], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "32e199dbb0be2ef56abd15bc74304ad7e3e51f31df7874c534935c082898c084:action", "state_id": "3c9018d3e09eccfd426d46ffaea353fc31eb4ec43ac8c31f9f76d96837ab1aa8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.49609375, 2.888671875, -1.046875, 0.90625], "student_probs": [0.07319322973489761, 0.8008559346199036, 0.015644731000065804, 0.11030609160661697], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1467f1f101b630d404059eaf79f2decbc9edcb9ad869a58ed3e6a1c6d25d6ec0:action", "state_id": "82ec46cdc583b205be6b57a56c6db22c0640fd2a69963c338b4dd06c60a967e9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.58203125, 3.107421875, -0.943359375, 0.845703125], "student_probs": [0.0666000097990036, 0.8322187662124634, 0.014487904496490955, 0.08669330179691315], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bd6660523fee368ae4dc614b0fab16507c0d2e32945f9da9c803a9fa1f91739e:action", "state_id": "fdd4c38ebbd29e622dafe91ac535c0d802215ac8496150db38127ec32d27869e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.328125, 2.9609375, -1.7578125, 0.171875], "student_probs": [0.033663444221019745, 0.9027764797210693, 0.008058479987084866, 0.05550163611769676], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "58a6f3fbd9c063a0e016aab18d8abdb9a53db92f87788ab98f0fb13e9ef0241f:action", "state_id": "8dec472c1d446c8a48da575f4e36bd8b867e15f7c3007f75ddc789d932ed5bdd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.13671875, 2.703125, -0.8828125, 0.85546875], "student_probs": [0.060858406126499176, 0.792312741279602, 0.021955521777272224, 0.1248733401298523], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ed73c88f9a2d07ba0097aafbab90e29e3ed45cc3e6756ac36156075476b81d45:action", "state_id": "fdaf26502d0d261776b24adf99265dae236121c7bfaff8a63f6e6410506e787c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.234375, 2.6875, -0.75439453125, 0.919921875], "student_probs": [0.0667489618062973, 0.7759310603141785, 0.024832893162965775, 0.13248713314533234], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "64f1b4b90713a76e30eca95f23622dd2f42c0f541c9be7b10396abb78e591b08:action", "state_id": "d9b1707a4bfc25d8ff71c98e0117922570c1f653b231a6027bc944c54b507a52", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.0390625, 2.916015625, -0.726470947265625, 0.806640625], "student_probs": [0.04341084882616997, 0.8336281776428223, 0.021830342710018158, 0.10113057494163513], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c1eff01e50a6696d5d7d2a5cb54e0783ca1a9bdd915f473fe9ea7380f3421848:action", "state_id": "99485e1c83d80d17847c03666b3a8d0da6d0e55fb02c046c471e7a549fbb3fa4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.62646484375, -3.6796875, -1.0234375, -0.63232421875], "student_probs": [0.368498295545578, 0.017395533621311188, 0.24776072800159454, 0.36634543538093567], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "52fb259d164282a45c2af22416ad30adf265336cf181001cd0c69c6f3e738273:action", "state_id": "3b498bca2cc4fa79202bca1e8e77cffd4f6b1e9bec013a82a9016a23fe48e46a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.657470703125, -3.078125, -1.0859375, -0.5958709716796875], "student_probs": [0.3566451668739319, 0.03169272094964981, 0.23235692083835602, 0.3793051838874817], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "50a765d206325f02be5945c924bcb6eb638942ee368d373d52f5b85b40697fcc:action", "state_id": "629b69ff8bb5125580f6ee8190a4bb677aaf7f851b82793e1623832ea66e77a0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.8671875, -1.63671875, -0.609619140625, -1.93359375], "student_probs": [0.060509681701660156, 0.2071145623922348, 0.5784613490104675, 0.15391448140144348], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7249164131f213e91e0de028877ad71ea938ecfd9fd31f3e7b5066d6c95ac659:action", "state_id": "e3a9c724892eeb7d2f21e12d3b5ba33ede4664fd128a976bc9055d44acf9e662", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9326171875, 0.34375, 0.06640625, 0.30078125], "student_probs": [0.093178391456604, 0.3339138627052307, 0.2530378997325897, 0.31986987590789795], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "810fdfdbf72e8ce926bfbd202bff31f3a7a454859e1fab4997846b38870f532e:action", "state_id": "a91145d6bccc82907e475bfa56f4b8ecf604454d7c6a7d232e740bf2bf89d6a4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.10546875, -4.0, -0.884765625, -0.70068359375], "student_probs": [0.2630707621574402, 0.01455437671393156, 0.32803693413734436, 0.394337922334671], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "20d64c6e18fbc3d3628ac312435ca008f1fdb9d98b904a70c729433cc0035a96:action", "state_id": "c663e78b3f2e1b5d8c8d20b577e2c85f80b18d10a98d83b157c0bf8ab3893a79", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7861328125, -3.5625, -0.767578125, -0.68212890625], "student_probs": [0.31342098116874695, 0.019514936953783035, 0.3192906975746155, 0.3477734327316284], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "07bd0c7ac4e85caebb4cc2c4fe3651a434ebb32cf29b3a01f945e3e8e6436915:action", "state_id": "59cbde6142123db56e822e4b5b43dc70cc11a8cb122ecc5fd4059447513cde95", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.55859375, 2.484375, -1.15625, -1.515625], "student_probs": [0.006141313351690769, 0.9514691829681396, 0.024962689727544785, 0.017426766455173492], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "84704ba628510c0f5b3cf3fb81a04280a0f89da76c9084f072afd4a699cbf473:action", "state_id": "1891249d1917780b7a6ace8adeefb2d1696989e4dc966bb7dc5dc02d7db9f49a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.96484375, 0.39453125, -0.990234375, -0.45703125], "student_probs": [0.053329624235630035, 0.5644585490226746, 0.1413305401802063, 0.24088133871555328], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "089f521ec6726f712441e33239a1f25f9f7c0f4766ffef0d57303ef79cc7b53b:action", "state_id": "f8a7f567130e218251d1c643a026c1ca5c4c3a05dbd51a0a637d6c162e47a247", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.96875, 1.0625, -1.08203125, -0.552978515625], "student_probs": [0.0353732630610466, 0.733044445514679, 0.08585631102323532, 0.14572595059871674], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fe0bbd8a5a872fdb488011a7c25245dab0f39dbb3c2231ab0f210becbe608506:action", "state_id": "6a71b0d5b9cb74e987af9aa57da93fab677489c70feb80b80173f18f1e42cdf7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6796875, 1.494140625, -0.654296875, -0.21484375], "student_probs": [0.03123648278415203, 0.7465143799781799, 0.08709307760000229, 0.13515612483024597], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "84f4ff9f8b0046dc9b762ad91a1f4fd4db87d38b199c6200054c4a5fc54ffc15:action", "state_id": "2ea65520e867513a6c9b21b7205568ffed797c4efae1e15a0bf26404599f086c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8828125, 3.056640625, -1.037109375, -0.7568359375], "student_probs": [0.006844314280897379, 0.9561084508895874, 0.015944616869091988, 0.02110256813466549], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "73411554a84ad901c682213c6de89f1ef3cc459bd49f3d141e84750fb36218c9:action", "state_id": "3eed9b26941f5e5db66cbc60aacb14b99ecf9fa327f29f2458c4eff3bb8d2104", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8671875, 3.09375, -1.11328125, -0.7548828125], "student_probs": [0.006716178264468908, 0.9585835933685303, 0.014273797161877155, 0.02042631432414055], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a31a93f6d1fa79e487e7258527a01a5b7d3735bb983d562d044102b404c136dc:action", "state_id": "fe262b8c9f9cdb00cc120c0f6548f7734dedc9b14a58230f7bb9936a972d3c0b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.84375, -4.4921875, -1.26953125, -1.54296875], "student_probs": [0.10318336635828018, 0.019847344607114792, 0.49806293845176697, 0.3789063096046448], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "20abf59e1053c30d2410336cd14fc4b88c797b80006eb7c69bccd7fd3c05a878:action", "state_id": "22f816f5fd36464dc704a8e44a0b5d88d253f94aebc95365283bac56db090db3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.99609375, -3.2578125, -0.984375, -1.041015625], "student_probs": [0.15077579021453857, 0.04269471392035484, 0.4146822988986969, 0.3918472230434418], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ad8d8b4d88f49ab0573eacea05df488688ae09ab704ba69e2c687562458ddc86:action", "state_id": "70e04afabba1a7e702d07af193bb9f79f69091a58b03c46504723bdd84d17788", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.88671875, 2.560546875, -0.4404296875, -0.87890625], "student_probs": [0.010708925314247608, 0.9144685864448547, 0.0454842671751976, 0.029338188469409943], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dda77f1f1f9ae2f075eb6910599a4312360be35eb5b4f2bb1e73332de52bbc3c:action", "state_id": "77b7540ba7fd081e4399f932047ce7ad00d7e5daecc391264cc7c82009cac2fe", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.60546875, 0.33203125, -0.462890625, -0.0546875], "student_probs": [0.06332573294639587, 0.4395676851272583, 0.19851602613925934, 0.2985904812812805], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c299b9e03dd591c4d6b4897b36152ce768a911855c27fdb6b615d20a87345a6b:action", "state_id": "5dcc385c550124dab394a3dbe6bb6481068e5be8f3ac2a4cc1f725ef6f328f13", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6796875, 0.984375, -0.5732421875, -0.10546875], "student_probs": [0.043094076216220856, 0.6185932755470276, 0.13029886782169342, 0.20801375806331635], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c637cc7e6620d5ebaebc06f156fa3a4b2c7d5ad3644a8aeedc39cc00109549dc:action", "state_id": "6f42585b78beec13b9da240323d57c899585f414186f87715e01e49c012480e9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.61328125, 1.466796875, -0.615478515625, -0.140625], "student_probs": [0.0335196778178215, 0.7293916344642639, 0.09091594815254211, 0.14617271721363068], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "78a9dd646d359b74635267bda7b8f27b9a9fc42b4afa99f4ce51870f320d3ad6:action", "state_id": "3284eb0584dec1845dbd79404d0d56f686bf837b015442611846e8455a8049f5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.98046875, 3.048828125, -1.2734375, -1.091796875], "student_probs": [0.006317703519016504, 0.9655062556266785, 0.012812060303986073, 0.015364008024334908], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f60506fc3ae824ff4cac81e9337e75c86d8ace0902c6d98beac9f9b7b40d75e6:action", "state_id": "be50636292919f399ee9b8daf857e8c645cc6b06a2d99f70759c3dfecdf422e7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.171875, 3.046875, -1.44140625, -1.3125], "student_probs": [0.0052592577412724495, 0.971401572227478, 0.01091850083321333, 0.012420706450939178], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2c19980d818903132d1757705a5970add3306e84811364df889c06d3fa2bc583:action", "state_id": "e9cdd8d48d8998055fbedf47f9c047b92d0da5eb01520cb058e58c900be8b640", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.25, 3.0546875, -1.51953125, -1.4609375], "student_probs": [0.0048413146287202835, 0.9744505286216736, 0.010050828568637371, 0.010657339356839657], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fd20b431dda9854c1e06a39cc2678635193b56855b569c88ebdf90aa404397a2:action", "state_id": "76e79b74d87b56c025748082e1438eaaf4dcf74b7f1720753cef77270cfa405e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.91015625, 3.04296875, -1.6640625, -1.484375], "student_probs": [0.006876319646835327, 0.9738025665283203, 0.00879494659602642, 0.010526173748075962], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "13a958319394663e507c84b697d5edc4b014390ab99bc881241f609e5e984579:action", "state_id": "1760a87cfa2e23afaad1abed67b5c4d7f1a6b2a4b585fc93d622ffd7ca460b35", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.04296875, 3.08203125, -1.66796875, -1.50390625], "student_probs": [0.005802365019917488, 0.9758077263832092, 0.008442390710115433, 0.009947567246854305], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9f3cd630c7c5bd883ddd9cfbacbb122a41df0c81e9ccbf646c7a0bf1a8c092e4:action", "state_id": "6719d0018c9c9bf9fdff430476ee1bd184acd164de49a39afc347dec225def55", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9765625, 3.181640625, -1.52734375, -1.34765625], "student_probs": [0.005608701147139072, 0.9750826954841614, 0.008789325132966042, 0.010519444942474365], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3c2bf86605944f617feffdd76d9104dfc7d033cb6a27a0452b3380c5732dd173:action", "state_id": "108e4ec9d3b737755654b4b7ae63d82e97c754bf6f61c0321c8aecb1b46fe6bc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.953125, 3.146484375, -1.74609375, -1.4921875], "student_probs": [0.005960419308394194, 0.9772575497627258, 0.007331441156566143, 0.009450601413846016], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c9bcbb73408ff696b301fb403719cad99b981ab91e34e4d72e1ebdb3c76de79a:action", "state_id": "d68b562f507eb0e011d2a5b42a071eb26e93a66e39c58d200b472be229d4841d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9609375, 3.1484375, -1.58203125, -1.4453125], "student_probs": [0.005892674904316664, 0.9756315350532532, 0.008607348427176476, 0.009868372231721878], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "10853ad800d74f6ce5d7e7350f6d964000f3b01647c00b3397de1f270ec51147:action", "state_id": "513e04f1a83a91fb76a9d805f955e0cd76bff514b9c575d164786bdb75101e96", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.35546875, 3.064453125, -1.85546875, -1.71484375], "student_probs": [0.004340130370110273, 0.9802682399749756, 0.007155665196478367, 0.008236119523644447], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "50c75ed484a24b4cf7ca48fe24a3300347150200df5f9e27faa70d69257cbd96:action", "state_id": "8acbce309bf839934223dbeb31d43338011e8096e5cb397ee356a568fd249aec", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.34765625, 3.09765625, -1.68359375, -1.59765625], "student_probs": [0.004224233794957399, 0.9786267876625061, 0.00820628460496664, 0.008942702785134315], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4b8364939c6e90146a4b05c020b8247922c6fae25517c21a6fbb935eb96e5935:action", "state_id": "493c03c5a8953a568e2a7bdc52514a5d544dfd8205be6d30ecee2ca23c2ca412", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.40625, 3.171875, -1.6796875, -1.5703125], "student_probs": [0.0037043895572423935, 0.9800890684127808, 0.0076605286449193954, 0.008545937016606331], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "93d4a52ff1ed08cbbd4c3f5869ec3e5c4356ab02b380cfb131dca416432142e6:action", "state_id": "f655e6a490c9405353ab1eeb230b47a6dfcafc317ba811c66c2ba751052cb8ed", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.34765625, 3.158203125, -1.515625, -1.43359375], "student_probs": [0.003969477955251932, 0.9770071506500244, 0.009121788665652275, 0.009901607409119606], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3eda36baa1d48cdc1b4bf669de463ed9084ce185a10aada20cf0018a5c368652:action", "state_id": "0d4830c0d0d8c620b935fc9d9a1c403066ec18fd047aae0b9f4903e74bc3981e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2578125, 3.181640625, -1.4609375, -1.29296875], "student_probs": [0.0042344327084720135, 0.9752584099769592, 0.009394499473273754, 0.011112749576568604], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d6c3c38a424a974fbeccd73a749f8d50117eaef0f63ba9ede5d1f0e36143827e:action", "state_id": "26789e44f99a70f38db33f71a3811888b204b93f7aacf90c6b8e9d2c8966284e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2421875, 3.166015625, -1.484375, -1.2890625], "student_probs": [0.004367622546851635, 0.974984884262085, 0.009318776428699493, 0.011328751221299171], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "634d324ab0b2ae479f1c68df0a9c12dc1aef03f0fb010db0ddbd7cc39a3a4e1b:action", "state_id": "88e0374866f851f950cda20b71ac12b846aeb8ac1f6d52738b842c7c8c65695c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2578125, 3.18359375, -1.5078125, -1.26953125], "student_probs": [0.004227077588438988, 0.9754677414894104, 0.008948723785579205, 0.011356520466506481], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a45ae6dda8f0b6932b16ec6e55e55265eb7a05a8f9ebdd3e2b0cd9aa30817822:action", "state_id": "81e56c7a03cc61e496a5b5ab6070ca96d8b3186cc568595b29d117537ad75530", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2265625, 3.22265625, -1.30859375, -1.15625], "student_probs": [0.004184155259281397, 0.9731357097625732, 0.01047795545309782, 0.01220221258699894], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1e9cdfcf7245c1bcb0376c5f16ea0b19c42a687cc4385797e52c1bf3599a4354:action", "state_id": "f9c640cd3923b60fbcb10be7c37ece2098d5d2517753eaedcf7f511d297569c6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.234375, 3.205078125, -1.265625, -1.072265625], "student_probs": [0.00421678414568305, 0.9711937308311462, 0.011109746992588043, 0.013479664921760559], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3c440035ea9114f4caf35b7ecfb70b7c496668f079c2ed1fb6765a382d1a09fe:action", "state_id": "55d70600c93aec30a7b87ef31140bf06d4cd0b4ff8f103baab246b1335f432a2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2265625, 3.224609375, -1.2578125, -1.03515625], "student_probs": [0.0041674054227769375, 0.9711350202560425, 0.010979651473462582, 0.013717876747250557], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6a31294df63e937ceaf56536e809b53a894e3015769c04b7d7b5e8994dfae314:action", "state_id": "b2fcbf1d22fd5e3429ca85b7ee648cfc867ae4faf81a7857320521806569fd0d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.17578125, 3.2578125, -1.3359375, -1.080078125], "student_probs": [0.004250292666256428, 0.9731921553611755, 0.00984369870275259, 0.012713837437331676], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2722a5ca5445d04cc30aca0a1d4f4b14b9884dd199b5f31013d36fd793b8e762:action", "state_id": "5aa83505894b8da9deebd25db893fab5157086415c9b410fea6a2033736588e0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.07421875, 3.265625, -1.375, -1.1171875], "student_probs": [0.004670795984566212, 0.9737682342529297, 0.00939848367124796, 0.012162541039288044], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b84cfb2c63a0de368ca4560f4fe90afd6598b91743d1b7e4847be1f730daa1e5:action", "state_id": "983d96af858cb19f801fedee5c945fbf06b3bc9ec91043b8c685bc3130cb9af6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.99609375, 3.126953125, -1.2578125, -0.958984375], "student_probs": [0.005755085032433271, 0.9659680128097534, 0.012041573412716389, 0.016235386952757835], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0535adc9decd3f6e3091b0f50b337e9ab8626a03b7e14839e551fc048357078:action", "state_id": "e500b190fe1f80764c66608ee25b9fb82cba758646838604f15eb74ead8e0191", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.984375, 3.111328125, -1.1171875, -0.8271484375], "student_probs": [0.005886510480195284, 0.9613768458366394, 0.014011113904416561, 0.018725568428635597], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "77d830b325f57ca4e6ad8f6084386dfa1442b8ecaadf2b5b788e3ad0968323ee:action", "state_id": "2e0d6bb192684690f2500d3ad4af4b994bcc408c1087b0a997c8c4efcd201dc6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9140625, 3.1015625, -1.064453125, -0.728515625], "student_probs": [0.006354793440550566, 0.9579871892929077, 0.014862166717648506, 0.0207959096878767], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8843562525485d639cb6f1bac8e81aaa3fb13e5fc84c3f985807cf1e10520e34:action", "state_id": "6847c641758865a911cb8667352def97c579bd4cd25874294aaad8a4c94d0868", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.86328125, 3.103515625, -0.9609375, -0.6416015625], "student_probs": [0.006647851783782244, 0.9544073939323425, 0.01638944447040558, 0.02255537547171116], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4190435f7739261cc2386b6d7c75e70ef7652f38ce28674264cb9f6d3bbf3b16:action", "state_id": "6ea586287aa82134ac85cb0f8b0db59df3a6ec7af7f2f169076620e7ee3579a3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.84765625, 3.107421875, -0.93359375, -0.65673828125], "student_probs": [0.006725935265421867, 0.9543677568435669, 0.016777412965893745, 0.02212899923324585], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "db9acf3acf89854ca67456c43bd6d374e736dbaa605a271002bc58eefd4ecf0e:action", "state_id": "35b0af410232806a64a8aeea154f582ea31268c62450c0c2b178ac1ea518a470", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7734375, 3.001953125, -0.7001953125, -0.470703125], "student_probs": [0.007926391437649727, 0.9397262334823608, 0.023183485493063927, 0.029163921251893044], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b22f4b37c9635845b092dc4031fd578664cd67f6bb52ab654a32e2dc7c160090:action", "state_id": "fc8a885d66abf58b2756aa11c05f67d46d60e88c9cb00ac653e60192ce4e0762", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6953125, 3.01171875, -0.4189453125, -0.216796875], "student_probs": [0.008354708552360535, 0.9250580668449402, 0.029939912259578705, 0.03664734214544296], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "df55133725ccd627db060b2f8a66f4095ad930ce11df52b3d2a2864b2db0e7c2:action", "state_id": "9427030a1b02bb887b1b68eef124b5613361c4c2a84f79ead87df56b9cf052f8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5546875, 3.048828125, -0.27734375, 0.109375], "student_probs": [0.009115563705563545, 0.910049319267273, 0.03269842639565468, 0.04813673719763756], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9fae8d0c50f4d6f375f4e8c10755129c2f298ba4cff8daa29f30feabf1e10d5c:action", "state_id": "27ab8b9800ed236b615cfe77463b729dbbaab2bce7078f6fa06c862a6a4d5e1a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.60546875, 2.857421875, -0.2578125, 0.23828125], "student_probs": [0.010213830508291721, 0.8859258890151978, 0.03930685296654701, 0.0645533949136734], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f99a2322845fcfe092801303d450c814f9f6f0163dff33a7bc8d2d29a1c35b17:action", "state_id": "4a0a31ab9bc08601d6a0bf0ba2af06c022173ca28988cd037be25c156b5c7736", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.359375, 2.892578125, -0.0390625, 0.3828125], "student_probs": [0.01239205151796341, 0.8704483509063721, 0.04640316963195801, 0.07075638324022293], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e85994e9fa726a2ac9a43f805c368acf1861b8b128fd75b23d4a4fcf6024aa54:action", "state_id": "dbc5e78fdf0c2d3fa05759930cf0372e4910ce97ad79b20761b6baddca63c3da", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3828125, 2.873046875, 0.06640625, 0.515625], "student_probs": [0.012128174304962158, 0.8552472591400146, 0.05166342109441757, 0.08096109330654144], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b49932160536b35a8e0ce8331ff7d8759f8649ac559f2cc2f935aa94fc9afccb:action", "state_id": "b426af89eb006f0c27775729d0cdaab0cd48af9c853f6d8e137292c8419338d8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.27734375, 2.912109375, 0.10546875, 0.609375], "student_probs": [0.012891501188278198, 0.8506675362586975, 0.051386769860982895, 0.08505406230688095], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c99741db55e81b07b5a49f566a5ec7e9ab058d32398ff93d5a92c7ce21715552:action", "state_id": "569f8776712ac18170da2de90bb9e3a3ffa54ac7c175ebfb7e5d6c987ff844ce", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.34375, 2.861328125, -0.15625, 0.37890625], "student_probs": [0.013003208674490452, 0.8715509176254272, 0.042635880410671234, 0.07280993461608887], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81c5ac54b0037e66d64a4a88b0826e95168dca6fbb29a1fb39ae8b9d29ee5f02:action", "state_id": "f428ffa9fe0f6d429197d51a7b976e1ed8c45ba41b8f9136e66cc8b2ebe88578", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3671875, -4.3125, -0.962890625, -0.896484375], "student_probs": [0.10451614856719971, 0.014939804561436176, 0.4256589710712433, 0.4548850357532501], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5dc4aaf121b9c3dcc983c23f973819a712f95b81e0627d388d4498166cb4d487:action", "state_id": "6e12862d243906cab61fc69e6aa6f228d3c0eb94b184cf4ff0eff8ebb0330321", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2109375, -3.53125, -0.9140625, -0.7939453125], "student_probs": [0.2524436414241791, 0.02480079047381878, 0.33970001339912415, 0.38305559754371643], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c818a52e8700df49b93fc4a10f269bd8ab66305b56b5f3a51a1546991d275b14:action", "state_id": "8febc4b1db4f3f7cddcb09c414b7185d68360a6a0d83382d6d8e7e3cc69ad9ab", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.26171875, 2.65625, -1.197265625, -1.3828125], "student_probs": [0.018779095262289047, 0.9445541501045227, 0.020029323175549507, 0.016637355089187622], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a2fb7217256d4a08ac6d2c29d21280a365a37cc6185eaaa28adde3a41a4305c6:action", "state_id": "f4c1c9e08b2ac7d882928663ad58be80713af8f6e0b127bc62468f1bd99f79af", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.33984375, 0.736328125, -0.7724227905273438, 0.046875], "student_probs": [0.16516901552677155, 0.48451149463653564, 0.10716719925403595, 0.24315230548381805], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "98cd7e730c2d6f553c44d2c7c7ec3a23ec966dcc28763af8242c717c24839405:action", "state_id": "86319e90a5f15e1c583a15bf7156b917856f13213b014144d7e4f06ff3e7407f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.1015625, 1.60546875, -0.6796875, 0.078125], "student_probs": [0.12091366946697235, 0.6665452718734741, 0.06782642006874084, 0.14471471309661865], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4a76544217693d5038b2e4124738eb0d0fca6ef07de0fab5566c19715149f3ba:action", "state_id": "a6d72f6cecea4e239fb6dc76257232ff62aa990e7ee4baf52d2bebd79554e37f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.1171875, 2.318359375, -0.275390625, -0.07421875], "student_probs": [0.0866798534989357, 0.7832041382789612, 0.05853608250617981, 0.0715799629688263], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "54b6be20bd9f5e57e1191a59d40d532ca28742556347a83066e7f27677a3c042:action", "state_id": "b05d07686bafcf5a298be7c1e75191f7a8ff5de65a401924e45e693bd36661c0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.193359375, 3.5009765625, -0.8193359375, -0.4541015625], "student_probs": [0.023516088724136353, 0.9457902312278748, 0.012574969790875912, 0.018118664622306824], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1793afffc73e09c0a3da96f4207cc6346eb7f678e5415d36aa78506eee91245e:action", "state_id": "f1c69933b36d0c870f9ef8b71f17ee7e0fb978a100273cc4fb0f305d80c17237", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.639892578125, 3.328125, -1.23828125, -0.8154296875], "student_probs": [0.018093552440404892, 0.9567797780036926, 0.009945966303348541, 0.015180603601038456], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6324af1398ef854be15f7a69a2bce416cb1930db8d0c5619bc0ef6f034ee66cd:action", "state_id": "d897612efd33a33e94d44798b532a6a25dea55a9bfc85516657efe299b8ace26", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.48046875, 3.09375, -1.71875, -1.39453125], "student_probs": [0.010017029941082, 0.9711737036705017, 0.007893228903412819, 0.010915939696133137], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81ae09864bd89325355c3853843919c31aeb683a1e033d55e9cbe24f9858f0ee:action", "state_id": "54a86be453dbab7bbd39cde541286a76aa161056fe1a317dc2989089e541a36e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7578125, 3.1015625, -2.00390625, -1.5546875], "student_probs": [0.007578590884804726, 0.9772107005119324, 0.0059253135696053505, 0.00928548350930214], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9a95dbbc9d5c1c86ff386664a87d79b0a68aaabbbabebc96dc9c2fb107997f51:action", "state_id": "671b9b8b1d24d2f8b0d8ea9313fa8195ee2bf84036a546dd8620374c4287942d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0, 2.951171875, -1.85546875, -1.44140625], "student_probs": [0.006884956266731024, 0.9731231927871704, 0.007955552078783512, 0.012036366388201714], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2c333cde656b52f786c53aa28028ba937d5d828fc8fa7e01feabf1717a1ded59:action", "state_id": "0f11782cc354cf64fb3a4564a1c8ff67a37adbdccecfba174b4c5f4c0db3cee8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.87890625, 3.015625, -1.921875, -1.296875], "student_probs": [0.007283056620508432, 0.9727060198783875, 0.006976740900427103, 0.013034267351031303], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cc4cbababe34a2b1bc4e59c614634b47c349a25176473688d84fc6eebbae224e:action", "state_id": "fc2ad8488699497e1f1560160b51648ab264f5afbcec9a2e3a5ec2e2321c4be2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.671875, 2.98046875, -2.046875, -1.1875], "student_probs": [0.009247200563549995, 0.9693875908851624, 0.0063555012457072735, 0.015009686350822449], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dc1758e1e44814ed9c4c93c4c615fe0d39266cbdd4f9f9e11dd6f853f6c30cea:action", "state_id": "027ae1d1b121079235da586ff54bf32b3a94bd89149241552611c41bc56a1350", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.16015625, 3.03515625, -2.52734375, -1.546875], "student_probs": [0.005435855593532324, 0.9807615876197815, 0.0037653069011867046, 0.010037200525403023], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "75453e1d010365f17e02a9ba7c30ef06e60fc8852e4be087770a50531ec85b86:action", "state_id": "fa8e20eba61f2c1e95e3c9c3362fca35b902f644f5d5b721d11e0a5c574fb34d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.109375, 2.900390625, -2.2109375, -0.685546875], "student_probs": [0.01724313385784626, 0.9506820440292358, 0.005730779375880957, 0.026344042271375656], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4dab39de8fffbfebb1910b34a29e5854bdee67e8ca1972be2565df26802e6429:action", "state_id": "90b8066b6888427776ffed2f75ef65818b7e694febf8f005eb838f2759ed0fc4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3701171875, 2.83203125, -2.0234375, -0.064453125], "student_probs": [0.036853816360235214, 0.9060618877410889, 0.007054310757666826, 0.050030019134283066], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0e9b16afb6cb531a6c052076335c3f0a74fa6dd7c87f95d90a7b803e170dd2ab:action", "state_id": "2a89518886de93d504534d1f90385153dd0f8fc1f0a5460fc051ba6e5806bfed", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.640869140625, 3.017578125, -1.84765625, -0.310546875], "student_probs": [0.02410125359892845, 0.9351539015769958, 0.00721005629748106, 0.033534880727529526], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "34d126f66d2f5061be3521abdd8cbb6438c66012f5adedd757fce014391e3130:action", "state_id": "5ed4dec8526230b2cdb5458fea9da7535455528894707d933e07a5f187854d5c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.318359375, 3.0703125, -1.91796875, -0.3544921875], "student_probs": [0.03145339712500572, 0.9318565130233765, 0.006352812051773071, 0.03033718466758728], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bf097aba88b8a94743d24f82743a2ee20f0714c4befaab41a147ce06ac5c1b93:action", "state_id": "ba8599eead652695fa27d7c5a6d0c36f68f0f3cd9e7d86b945c250d77d1c9555", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.2734375, 3.173828125, -1.65234375, -0.3759765625], "student_probs": [0.02978958934545517, 0.9358213543891907, 0.007502623368054628, 0.026886383071541786], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e1ed65fcf46f023f2f7cb1469124e4440ebdf08ed37bdc4fd95e7de26df4dea1:action", "state_id": "aee2b64dccfb916e297804243c1e4157d46f2bf39debb9e26c45a9115a9647ae", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4677734375, 3.041015625, -2.2109375, -0.345703125], "student_probs": [0.0280013345181942, 0.9354626536369324, 0.004899279214441776, 0.03163684904575348], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cbd98238bf19cec18d31b28299cc90ee474aa744f804172303061109efc7825d:action", "state_id": "3a6d557ad5e8b016067faadd2cfc0f490bc0d3b6f19ede864badb94051e91478", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.36328125, 2.94921875, -1.55859375, 0.44921875], "student_probs": [0.06446705013513565, 0.8558470606803894, 0.009433613158762455, 0.0702522024512291], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "623c7c6ba0378f8c24344f7dd640f9d5abe22284e5b729d2d0a09c5e415a1f3d:action", "state_id": "3b9456f562bddf8fdb79b69611e4d5747491ac882521915a4f0e6d8c8871028b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.265625, 3.146484375, -1.48046875, 0.25390625], "student_probs": [0.050019022077322006, 0.8918185830116272, 0.008726022206246853, 0.04943627864122391], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ba0add295d1d10fc537217f8e0ecb21a46dd445b5483782e98956b3d14216bc6:action", "state_id": "a8b6f4514eaaa7eeaf865c039c621624507e0c791a92c22cc5ab316eb1875053", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.1953125, 3.234375, -1.515625, 0.19140625], "student_probs": [0.043360523879528046, 0.9056128859519958, 0.007835086435079575, 0.04319147393107414], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "93169ebf75c94f30bd3a2c01f5678bd49e32f4b41f9679747f0d375edffb7ba8:action", "state_id": "894458ba8116e675f7c672b3b64a700e6c58a53237048fa5cddc7fb45bc60e5f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.1328125, 3.1953125, -1.69140625, 0.1484375], "student_probs": [0.042448367923498154, 0.9075860381126404, 0.006848773919045925, 0.04311682656407356], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9ee9620a0067be315d4e302d15f9feea69f5a70895769e492d7b7f38ecb7db9c:action", "state_id": "f0ca10ae54b8849f50a6c9bacec87760370a65a9164df4cd0774dab3d12467f6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.03125, 3.17578125, -1.60546875, 0.21875], "student_probs": [0.03904779255390167, 0.9062521457672119, 0.007599386386573315, 0.047100625932216644], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "012b4cdcfc187ebd92f1a23df246a166c9b178082a8960bd5725d3ec67d73d12:action", "state_id": "300cb157669a1af79bff23fbeeb93f138414670c93f1d59b4f16ca259009ba56", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.4765625, 2.9111328125, -0.7347412109375, 1.009765625], "student_probs": [0.06938129663467407, 0.7917040586471558, 0.020662358030676842, 0.11825230717658997], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5f8825ba198e4824bb9ba5a81f6d87f8366d63307cc1728b13209fb6d4b0bea3:action", "state_id": "21da7591a54c9692268d75a86460e48fa287b99a04b2123aeb411f7558040557", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.58984375, 2.5400390625, -0.5615234375, 1.02734375], "student_probs": [0.1010601818561554, 0.7104591727256775, 0.031955648213624954, 0.15652507543563843], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ef5eac49fd7280dbfc0827c4a262ac0b26bbca0a77bfc49124d780bd011f28d6:action", "state_id": "71ad2a6f458c24b6978521cae2e437c4f87e06564851999a3f6b43977a5a021c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5673828125, -1.8125, -0.84521484375, -0.73193359375], "student_probs": [0.3455895781517029, 0.0994977205991745, 0.26175785064697266, 0.29315489530563354], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d8f21741068edc7e528bf72bbe0e3d9aa8744935fae5608f7b259daa4dd56824:action", "state_id": "0a94ef519599a67773433dd83222964f00921e9dd925326b9bd85d6963e40bc8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.66015625, -1.65234375, -0.4755859375, -0.767822265625], "student_probs": [0.2880687415599823, 0.10680574178695679, 0.3464607298374176, 0.2586648166179657], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2dec1d56201437f95e320614aa7ec0623761c76f076514937681a28607f70a1a:action", "state_id": "643e2230c3106daa7d6436a16583cdba8dc316afa59796f11355fa020e83034b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.740234375, -0.75732421875, 1.623046875, 0.36328125], "student_probs": [0.231090247631073, 0.05168924480676651, 0.5587045550346375, 0.15851594507694244], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3b9b604043512dd81964476dc23cd91f973472c4f88d4ac7bd6e504384b3fb2d:action", "state_id": "214bbff758341822c2c6fffbc67f2dfeae9825913e02b483c2c5f0a8639a7b8f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.33203125, 0.01953125, 1.8994140625, 1.11328125], "student_probs": [0.2606668770313263, 0.07015753537416458, 0.4597238600254059, 0.20945170521736145], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d157181a5ce516cae3b64dd1dbe7156416d6fa69bd733e5cbd314a6200ac4c4e:action", "state_id": "33b03ca07336cca862ea584ab63930ec71cfefb17db04ab7193a4d378dd44aad", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.5, 0.28125, 1.546875, 0.85546875], "student_probs": [0.16449785232543945, 0.13217774033546448, 0.4686107635498047, 0.23471365869045258], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d6b81d57d56206fcde74f9abca13aab41e698314752a5450ff5c9782a1ef5db5:action", "state_id": "126099083b1d5c8b4a784756f516de7740d9f49fb126fa6bf7583c82c1de7f93", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.2109375, -2.1484375, 0.29296875, -0.14453125], "student_probs": [0.25853830575942993, 0.037245973944664, 0.4279259145259857, 0.2762897312641144], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f976df0acc15dea5d28db38e28840fbd64f5d54360757ff5d0c59ad69bc086a9:action", "state_id": "48cdba58f09ef226cde7e51311795d912af5d7c70c56eb98eccb9bd42016d18c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.318359375, -1.79296875, -0.771240234375, -0.279296875], "student_probs": [0.34429365396499634, 0.0787978395819664, 0.21889980137348175, 0.3580087423324585], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9791d0398724d96e0388c7b4cc58e99ff908d3f1f099000019e3d0894af0a489:action", "state_id": "62fab2723b1b91369de6e3aff9bc00f04eaa69883ccff68ac92fe5318e4e9060", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.349609375, 0.2265625, 1.1953125, -0.322265625], "student_probs": [0.11772266030311584, 0.20945385098457336, 0.5518373847007751, 0.12098605185747147], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fe1411231483b1c466f99847bec3e10425ed48e99c962a072e7140cbfa7e6e73:action", "state_id": "b873c3f87faf8f3a4f804346588bea339358a1d4d3f8798931b282a03efcc127", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.890625, -0.66259765625, -0.078125, 0.80859375], "student_probs": [0.3980312943458557, 0.08420952409505844, 0.15107564628124237, 0.36668363213539124], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "509fa79445c1ff3b7528a46a0e0ea1b5ed105de96bab771f23527595c61c31f6:action", "state_id": "9ef2487caeaa4bb37bb622243cfc0bbcbdc405d7cd5297e391ad31eac15f6bc5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5703125, -0.224609375, -0.681640625, 0.45703125], "student_probs": [0.1639004647731781, 0.23158857226371765, 0.14663276076316833, 0.45787814259529114], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7178741f4ebd130fcc4aca336daa070b5aeaca5c3af11e8950b342075188ac61:action", "state_id": "c00930756d1c3d641f6048b904b96b15eca9c1057e400523cbf4aa2a5cd83883", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.150390625, -3.01953125, -1.158203125, -0.3642578125], "student_probs": [0.23034223914146423, 0.03553171083331108, 0.22854970395565033, 0.5055763125419617], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c25d69eab4166838a4ccfe576197d84185f9dd004827274b356f82c01e9b6644:action", "state_id": "9c3db393274953dd20123016b0a5f4b3de8bf9e5d2ce403dfe1ddf694847890a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.966796875, -3.33203125, -1.07421875, -0.60028076171875], "student_probs": [0.29114001989364624, 0.027345988899469376, 0.2614864110946655, 0.42002758383750916], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9f839872ec1b276b55db5bcb9a44cce7039405e7e4d100dbf30fa19d43053154:action", "state_id": "ac76756b8fcee681b7eb1cf00b9e864ce4a1f36831261af8ce684dfb77d661b5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.5078125, 2.2421875, -1.82421875, -2.34375], "student_probs": [0.008351179771125317, 0.965265154838562, 0.0165435541421175, 0.009840094484388828], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cdbef1dcd0b3bc519bee7a59197e6fb123c9f56e55fe544c3ff817c9e06ec3a4:action", "state_id": "3623054d02835d0430c77591bb3583c98e8b6b8bbad6a3bca9cac0ff4c12eb36", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.56640625, 0.7578125, -1.041015625, -0.513671875], "student_probs": [0.06339028477668762, 0.6477658748626709, 0.10720053315162659, 0.1816433072090149], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2d50113328f04ad9b16270f49a0cc0710ad10194c5a388bdb6f14358a21a1546:action", "state_id": "a6ae79dc1f152e855b4afacbcfcdf620707b85396aac40250bf6a7c7dc2a1e82", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.11328125, 1.30859375, -1.58984375, -0.978515625], "student_probs": [0.027453633025288582, 0.8408165574073792, 0.04633678123354912, 0.08539300411939621], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b64ee171de313ad06f641bd4866baa4d477100250907b67c3ef3ae0d5d055111:action", "state_id": "b59f1aad88a57ec9e83da0cc6bebbe5436246b7cd8d97b13cf4321ed512cbe8a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.60546875, 1.794921875, -1.35546875, -0.8974609375], "student_probs": [0.029163213446736336, 0.8741908669471741, 0.03744630888104439, 0.05919966846704483], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3bc3c8b734c2115e1dfc0e2ca539ff40c835f99c55c106134f111b93152108be:action", "state_id": "2fed2356514ee850e42fc2bbb6104621f64e5b288865a8693a4e6fc53a19d781", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6171875, 3.046875, -1.71484375, -1.056640625], "student_probs": [0.009113719686865807, 0.9666566848754883, 0.008265784941613674, 0.01596386544406414], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c8ceb0ff2cba0ab43b66a0ccb2121fb2df37dbc2aca3c12b06c357ca2d6d2386:action", "state_id": "4fcf2fffc7c05f30665bbe1318ec675032443d90b159007e26d59fefd2e208b2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9453125, 3.04296875, -2.00390625, -1.359375], "student_probs": [0.006647881120443344, 0.9751384258270264, 0.006269549019634724, 0.011944078840315342], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1d0e55736a6382fbab5cc8cdc127399599a7f27f904fcc50b85292479bd81ca0:action", "state_id": "b67e0cd3eb8769209a9dbf9931fb0a0b5e3874163f6e305500000f6f6b806c68", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9140625, 2.849609375, -2.0625, -0.501953125], "student_probs": [0.02177058346569538, 0.9384517669677734, 0.006904145702719688, 0.032873570919036865], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5c3f245bc3f3194ae6c35ab28ec3480147a2e48f62bf5db447e4eca87b1a696b:action", "state_id": "c186f3bdcb60d72b390cb0166a7f186b6d5344487949140b0cb214fcb40b1ea8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8212890625, -3.42578125, -2.55859375, -0.8759765625], "student_probs": [0.45522502064704895, 0.033659644424915314, 0.08011692762374878, 0.43099838495254517], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3bc7f502d72b5a8d1c6e6d9158ec411b6f55bf31a647010786fae5f72d5d3ba0:action", "state_id": "39281348c68f1efcfe7a12e28896777d300d0feeff44c6f8a4abb5f3490cbdb9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.453125, -2.7421875, -0.939453125, -0.58721923828125], "student_probs": [0.3859887719154358, 0.03912438079714775, 0.23733678460121155, 0.3375501334667206], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2ae8a67951a5ca4c816d642fa463ee6b982fd776789eeabd20e63999f4946f0f:action", "state_id": "a5adc25a0e0abf106d458f25eddde4a8e2b017c3cdbf3252d7539db39dcdafab", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.005859375, 2.08984375, -2.80078125, -1.0234375], "student_probs": [0.0412345826625824, 0.9113986492156982, 0.00685073109343648, 0.04051608964800835], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f7718f22da87fff7aaf5b49e6b5d0f1c25a883b9b3dad4293fc63167fcfb684c:action", "state_id": "735e6d5bb07f3f515fd6d2a902e049591ea255831e925440f685ad3db83ec648", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.125, 0.9765625, -1.36328125, 0.61328125], "student_probs": [0.1923602819442749, 0.4507588744163513, 0.043427322059869766, 0.3134535849094391], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e812b50e0b13569bc37ad50bf6ba73b78536e8a91649fb1ba6c02b6173106fdc:action", "state_id": "fd60fd81dfe52bce28f8f154a7f20055128b9ca982af92b92b825c907b403e69", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.013671875, 1.33203125, -1.953125, -0.03515625], "student_probs": [0.06900379061698914, 0.7204417586326599, 0.02696954645216465, 0.18358486890792847], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "35dc6f9d105409e8d0b9a714a2ca175d82ecf95e58bcaf18c072dab56160a078:action", "state_id": "128b0c9efb691ea5c21a3b43d7bc2e02082cccfe8812fc174c4668206a2aa8d9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2890625, 1.837890625, -1.7109375, -0.32421875], "student_probs": [0.03692137077450752, 0.84196937084198, 0.024213625118136406, 0.09689561277627945], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8fb7891c2a1ba123a74ca7c42ad2ee75b26bbd70faa3cc2bbde7a86455e012c5:action", "state_id": "b9ba2e8ad6451f10da2d4d1099a3884742fd5432386b887a2c382a505b9c940d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5, 3.05078125, -2.015625, -0.892578125], "student_probs": [0.010189610533416271, 0.965021014213562, 0.0060844942927360535, 0.018704993650317192], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e52616078c69759c9e21320896c1530e5fdf5b2038268729add595212f75fbce:action", "state_id": "2807b48c6a36697e2ae1204cfdbecc76c60857d58c27c1dd56990d86ff4fb0db", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.15625, 2.98046875, -2.45703125, -1.54296875], "student_probs": [0.005755620542913675, 0.9793563485145569, 0.00426053861156106, 0.01062763947993517], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8a7d2f444e77d664b7768d6ac22766af76cdfc6b8ba24849894597a191c165c3:action", "state_id": "599881dcfe8d1034a86bc49a92930ba14d2c47e5f71d0f6fb1ebf633c31de720", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.27734375, 2.990234375, -2.20703125, -1.78125], "student_probs": [0.005059171933680773, 0.9812045097351074, 0.0054276990704238415, 0.008308645337820053], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4f0b392c53cd86e2631beeba74f61ccf81f1a43c1b6324dd0c9744883f81e1b5:action", "state_id": "16a493094d66b47150bc1b1ee50c0a35226cb95572a9b8caa50e44919df2ad98", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.29296875, 3.037109375, -2.12109375, -1.73046875], "student_probs": [0.004752926528453827, 0.9812612533569336, 0.005644240416586399, 0.008341646753251553], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "32e01cb8eea83ffc50fb4a25d40a745e5c69ad5e520dad7c980fdbfbb5c791a4:action", "state_id": "10f3f237b2530836d961f89a2d797954db755b6d220ccdfc448da041dcd04eb1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3203125, 3.1171875, -2.03125, -1.8203125], "student_probs": [0.00427623325958848, 0.982964038848877, 0.005709520541131496, 0.007050316780805588], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9efb4f5f7d3a1bfca5a4cc4f08f13a4b196306573c523878388113ad818e46e6:action", "state_id": "8eb4e9c752a02522d3ed6160218f9e40f96aac2e876bc6deb5ebd68a15e40440", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.4921875, 3.0390625, -1.9921875, -1.70703125], "student_probs": [0.0038865020032972097, 0.9811835289001465, 0.006407758221030235, 0.008522125892341137], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5f8e7d59ff288fc6b914bb14fcc6462dcccab0c972de4c63da1ba3d27c054257:action", "state_id": "e22e845335641aba4f2af4dbb5288660aa7e6a7f59c3535ba91cc37e256b7c52", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.50390625, 2.6875, -1.3359375, -0.744140625], "student_probs": [0.014197208918631077, 0.9386584162712097, 0.0167938731610775, 0.030350441113114357], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8c2be45e3211140a08f4e864bb82806d3c29f4436630252335914f251a5d12bf:action", "state_id": "35b5efa225e10d3e89a2be6d8ad3832d8c069fc5a78b78f5ec4afa0e8b56e021", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3671875, 2.771484375, -1.080078125, -0.5888671875], "student_probs": [0.014874345622956753, 0.9329110383987427, 0.019821107387542725, 0.03239351883530617], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "46064e44a0a3d7c3669a52f8ab489f0caba9a726e3a8e0ee91bb428613156f64:action", "state_id": "007fd581adc3d7004374818e957ef3565da2d2565e67491cd566d499eea0d004", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.39453125, 2.8515625, -1.55078125, -0.7080078125], "student_probs": [0.013573264703154564, 0.9478495717048645, 0.011609828099608421, 0.026967313140630722], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f54850bb3e338074fa973f23abcbf51d208b17390cfd2b462eb4dbf3581bc352:action", "state_id": "8b0087773985f99ebbc17126a29e6bb16bc41cf3d462eeb69c6d00c3be3b7686", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7421875, 2.951171875, -1.62890625, -1.15625], "student_probs": [0.008838911540806293, 0.9653812646865845, 0.009899111464619637, 0.0158806461840868], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "766767eeb2629200121ba705e6cd3b6cb78382f4749a8c6dd021d128fd438664:action", "state_id": "a0b2cdfe40ee066060163e1a7453ad7caf13591d1048c9656ae8f0c7e7f78079", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.47265625, 2.98828125, -2.2109375, -1.828125], "student_probs": [0.004174978472292423, 0.9824473261833191, 0.005423969589173794, 0.007953725755214691], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e2f004a9e3e3a1f4a6f479a723a61264fafa592600889d76cb615cd2a5e2bd47:action", "state_id": "cd1215e94373b3535aa6bce1b4d924de04cb0a3ad12705156ae09498daec9232", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0078125, 2.939453125, -2.109375, -1.6328125], "student_probs": [0.006937320344150066, 0.9767016172409058, 0.006267346441745758, 0.010093742050230503], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6bf5e07f645061f55992036f8f4909a5441e0e0c43bc8e314d330361547df899:action", "state_id": "795308d57d89fe8671ad08edf576b45e3cf5281f56da83567bb5ce6312b5801e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.64453125, 2.978515625, -1.087890625, -0.8251953125], "student_probs": [0.009361757896840572, 0.9530620574951172, 0.016334407031536102, 0.02124175988137722], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7192d2cce54be3b8a947a903df38d13bbe7f98ac31e5fa9548fdc2459572b385:action", "state_id": "a99b1ac2df4aed2d9add894c40b0e46a9eddf0ddfda365844caf0f83d5a26ea5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.09765625, 3.072265625, -1.66796875, -1.1484375], "student_probs": [0.005524203646928072, 0.9717134833335876, 0.00848946999758482, 0.014272831380367279], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4ec4da4ae0c34f56cdee47bbcf2c35092a97f1d98d672137e7a6569dcb750ae5:action", "state_id": "8a4520e8a9946b72a63abb94a648b5899362eaaf69ba260f4b937e77c3874513", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.01171875, 3.126953125, -1.69921875, -1.38671875], "student_probs": [0.005723303183913231, 0.9757612943649292, 0.007822828367352486, 0.010692537762224674], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dcf5a7cd02fc6233d446fe2c96d234a4bd3eea62d609ac8231f46d17d088e304:action", "state_id": "704104676c0123ca49e7d8e226863561af01d8c9445e7ff9c96dcc8129971ec4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.03515625, 3.20703125, -1.84375, -1.51171875], "student_probs": [0.005181829445064068, 0.9797973036766052, 0.0062749432399868965, 0.008745993487536907], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c5f820c3d802a8a3e5f35de64819f5d1bb956f1c7d7feae1c65ec55276e6d549:action", "state_id": "5faac44f40a4f73d05c42c7d04e7cf95a82f74eefb90554a983d3f4d86987a24", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7529296875, 2.8828125, -1.91015625, -0.2578125], "student_probs": [0.02445882372558117, 0.9277230501174927, 0.007688798010349274, 0.04012935981154442], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "47f6790d36b71b40257104015ec6fff6f728e698c55c02805c92091df7adf2ab:action", "state_id": "0dbe9e8ae3e0296bff8a9a73385136c961ebfaa3f4ce0baa2e046fbab8249867", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.939453125, 3.03515625, -1.48828125, -0.4521484375], "student_probs": [0.017719542607665062, 0.9431992173194885, 0.010235274210572243, 0.02884604223072529], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "92353eddca34ee197a931271e798f979309e3a17d59393dd69e7408d1e23886d:action", "state_id": "24ba92de1d670b178490c1b2e44dc784cbe40036d37a64b7a33cddbc71e77fe9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.06640625, 3.123046875, -1.2265625, -0.5751953125], "student_probs": [0.014394081197679043, 0.9498178362846375, 0.012263910844922066, 0.02352416142821312], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c7263a4d47a4b3763185909786fd93435458dd1f3c5be3c660fdab9aacd5fb73:action", "state_id": "188324fa912a8882ee0542c5392bc65d2b8eba6e09c4bcc979f1d5a84cf764cc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5185546875, 3.2158203125, -0.774169921875, -0.13671875], "student_probs": [0.02217232622206211, 0.9281746745109558, 0.017171133309602737, 0.032481830567121506], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6cc4e606fc072d3df55f8bd765523bbf4d40357d71b9b15e2f1b3077dbbed8b2:action", "state_id": "8b5bba2c315b23e78f2678d7e901f5b27db155ab7ba7f558f3d0a8a4017b9bc4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.845703125, 2.892578125, -1.58984375, -0.4130859375], "student_probs": [0.02220143750309944, 0.9330308437347412, 0.010548845864832401, 0.034218765795230865], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "56ca48f632822a09d1a0121a47083c6503f9a088a0a283c61fdadc0ed797993a:action", "state_id": "4696ee52cda05ee05824cc2a7218d69600258ca98f8acfd9aa055b13eac6d39a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3359375, 2.529296875, -1.5234375, 0.890625], "student_probs": [0.08429964631795883, 0.7557699680328369, 0.01313135214149952, 0.14679913222789764], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "00de7cd7e422ad1de6ba8f7423962314c190fdd0dad8d82744a6ab4fd1ae5deb:action", "state_id": "0d786fd73d3c4ab221eda0af4326dde3a17d95b3785f735e3b63142fcd1b3e41", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.212890625, 2.69140625, -1.98046875, 0.4296875], "student_probs": [0.04689435660839081, 0.8559347987174988, 0.00800702441483736, 0.08916383236646652], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d3f8c0ac25662fefb54eb099f5e77c6f6eaaf8211503ab45409fd422de52bfb8:action", "state_id": "c952b1d047904ceba1270c574cbafa6c3f66f9765e771c568bbc51fa57e0e249", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.59423828125, -3.8125, -1.91015625, -0.660888671875], "student_probs": [0.4456775188446045, 0.017838051542639732, 0.11954318732023239, 0.41694122552871704], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "08c4472005020fbffc14d35961bce2a2b9f8584401a9f48e853ff2c0b0fb64ed:action", "state_id": "0444284cc956b7eb1e748f19dc1e1dd86ff83c629d8788cd63afd86e8a00eab3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4267578125, -3.2109375, -0.8115234375, -0.624267578125], "student_probs": [0.39014309644699097, 0.024102943018078804, 0.2655353546142578, 0.320218563079834], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "79c949d24493047937d47e641ae6aa834839161dae71144ae3373bea00728abc:action", "state_id": "f0c8d339bdf7bfb915b58a12f50487c6d0550cab194226317a6e8860c2b6bc1a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.515625, 0.37890625, 1.8486328125, 0.15234375], "student_probs": [0.15723173320293427, 0.1371399462223053, 0.5962908267974854, 0.10933750122785568], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "34cd8a3725f5f6ef48dbe4bab2f78b0a971df7ea653785e430940e9a9fa593aa:action", "state_id": "f8248be5a1f8eaee1262fd50f2609a98dd8563dfadc77a635f91a443f5980787", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.466796875, 0.0390625, 0.14453125, 1.109375], "student_probs": [0.45333799719810486, 0.1087338998913765, 0.12082851678133011, 0.3170996308326721], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "035690c8ef2784bb69caeebcac70001291f7582658ce80ccdf0c79cea09d816d:action", "state_id": "c08c1705429a5f1e47e86a9dd4f92cfa9a3686e5cdf0659e97d1b2a971d2868c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.30859375, 0.171875, 0.09375, 1.408203125], "student_probs": [0.3673275411128998, 0.1178644448518753, 0.10900679230690002, 0.4058011770248413], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "415358d6bf1c68fe6aab378e13a9cd187005063996ee3cd240bf516caa4ddfd4:action", "state_id": "458c6f674ce5253aa739f09febff9c845f6d07d4a09f9969aad2b3f47cff724a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.34375, -0.201171875, -0.310546875, 0.69921875], "student_probs": [0.16598522663116455, 0.19142132997512817, 0.1715889722108841, 0.471004456281662], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "43f37a78db7567b9bc0d3eef8e56201f4718f9eb89257193258833191ba72f9d:action", "state_id": "be09788e16242b30ddc4bedaddb89d3ce388f195830db0df1929f7ff719511e6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8828125, -3.21875, -0.9453125, -0.3525390625], "student_probs": [0.267699658870697, 0.025891847908496857, 0.2514805495738983, 0.4549279808998108], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aee9601640720a43b0b3afc57b1d570fd0782f40cefb314291c12bf7134250e1:action", "state_id": "b26a7a84f9a219826ee207533caed05b5fd9827cc636fa38484dd6de3fc9071d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.51953125, -2.73828125, -0.884765625, -0.77734375], "student_probs": [0.18929696083068848, 0.05595608055591583, 0.35712385177612305, 0.39762309193611145], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8b8e73dd5365f8c79807bfe4433c6e42c48b1ff1d9a683e3251a5ad76c5b9e68:action", "state_id": "3d068a2528163d57efa8fe6d7af156b0d421cc3f5b2d1089de8525c7b846c2b8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9609375, 2.572265625, -0.3720703125, -0.81640625], "student_probs": [0.009794767946004868, 0.9114634394645691, 0.04797670245170593, 0.03076506033539772], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5e54fdd16b194b81af063c307fb3fb80e32e7bb9540f826667bb3efea3a9fb1d:action", "state_id": "dae4110d5b8bd680b398ab5f72b12ebaf6e8e07b81ec8501c501fe1eec754d97", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58984375, 0.4921875, -0.59130859375, -0.0234375], "student_probs": [0.06051637604832649, 0.4853864014148712, 0.16425977647304535, 0.28983744978904724], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fc6b9077f8aac2d46fc49c83dde6419490b844c4ed26a1939ab7bf1fdeb2514b:action", "state_id": "4377bc42c61bdb064b4cabbd6bc28d85ffe26983fc8a8f66f08ca109c44d03a7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.80078125, 1.25390625, -0.7509765625, -0.255859375], "student_probs": [0.0336029939353466, 0.7128725051879883, 0.09600687026977539, 0.15751756727695465], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "468b509359d8a4901321790cfa7e125bb2868dc957c99ef726ca3d7eb7ad5ad0:action", "state_id": "c8647e9ef46458d55b6cd04c6d5fe429fdf1ecf9ab964139c1ed582a1545050b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.73828125, 1.623046875, -0.7568359375, -0.333984375], "student_probs": [0.027345996350049973, 0.7883153557777405, 0.07296758890151978, 0.1113710030913353], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f56c51cecc010a4b0a77d9e70f6b431a722aafc14138bcf0e8348734f60081e9:action", "state_id": "e470b3138f0d37a1e4ede058650af609de97546881660186c06ee7ba1d4c0fa7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0859375, 3.041015625, -1.359375, -1.19140625], "student_probs": [0.005746562499552965, 0.9683125615119934, 0.011883659288287163, 0.014057176187634468], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7d749bec139e4e4b19b82272db212e064d3afa3e7e80a28cb9880842b396b34a:action", "state_id": "df5734dcd175fcc0b81c06acb1b236fbe38f4e82483d229b27db18dd2cfdc532", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.28125, 3.056640625, -1.51953125, -1.47265625], "student_probs": [0.004684717394411564, 0.974764883518219, 0.010034453123807907, 0.010516015812754631], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "180251dd55cccb4202a99ef62fd939d576979d0081c5394692655b4ffbfb8e27:action", "state_id": "406f35d9be8d79d17c318164075e3c4413259b4ef36ae7b3e1d74d7ab85fbd5d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.35546875, 3.08203125, -1.6328125, -1.59765625], "student_probs": [0.004254227504134178, 0.9779056906700134, 0.008763273246586323, 0.00907683651894331], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7e084973edbfb7435061cdd613e5fd1fe211f69be87cd2b6768997f387ff6136:action", "state_id": "785359f1a2320860fbb0bd89e84a6c48c1b9bb47a6081049c6d56829a66510d2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.4296875, 3.07421875, -1.703125, -1.65625], "student_probs": [0.003985892981290817, 0.9791331887245178, 0.008242667652666569, 0.008638241328299046], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4702cebd67b93918b68b9ed51e4055844e8839ab8ccc35ec31430c5ff30835e8:action", "state_id": "c19613be672c6906f595521208fab8684e5c651c8acefa5647db83b68108cea7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3515625, 3.08203125, -1.5390625, -1.4765625], "student_probs": [0.00426215585321188, 0.9759085178375244, 0.009604915976524353, 0.010224380530416965], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1bc4d3b02b34fb7ca93bf684175fef88a9b88ce145e42a592bd672fd67c1ffc2:action", "state_id": "0f91da507ee26461598ea4786d41030ce27ac07ffe9c8502e5b4a845b510b5c6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.38671875, 3.140625, -1.67578125, -1.46484375], "student_probs": [0.0038906726986169815, 0.978407084941864, 0.007921015843749046, 0.00978115014731884], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dde44bc09f22c0e270f2981b9e99afd5353606b2598145d6e3ea16a56faf712b:action", "state_id": "2e86310c7f7ddab0d24e8d99f336345605b51b11033cff0be1b6843beb04bd84", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.4296875, 3.07421875, -1.75, -1.5], "student_probs": [0.003981579560786486, 0.9780735373497009, 0.007856694981455803, 0.010088196955621243], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "35e65f2f643ae8b8e0fdc0cd17d8c996c5adf99519ae0a73b8de4283f8896cd2:action", "state_id": "b607d900c671dd6a15956d030c01e489ce5c396070e0f6b2de28429db29bc131", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.375, 3.05859375, -1.609375, -1.4921875], "student_probs": [0.004263689741492271, 0.976259708404541, 0.009168373420834541, 0.0103082787245512], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9d44ffd5ac4046500be371887ea2999f67c8c7bbaeb5f02198ba0972270a480c:action", "state_id": "eabb11f235eb5f64e53f618f4024c740f380a40bcb7ba2377190733d1b35181f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3203125, 3.1171875, -1.35546875, -1.1640625], "student_probs": [0.004225307609885931, 0.9712579250335693, 0.011088802479207516, 0.013427999801933765], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e103bd6442882e2293190894dafeb875097a7490d87659fbf207136e603968c5:action", "state_id": "94aeaece1454bbf4d71641b6a53b6e351f20979cf0e57a3df8743723b76dd9ab", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.21484375, 3.12890625, -1.203125, -0.947265625], "student_probs": [0.004616834223270416, 0.9662853479385376, 0.012697789818048477, 0.01640009693801403], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0a5e5e5211d2c258f47764bd7e71dbc173e86a21ab9e16ba786c815dc782b961:action", "state_id": "6eb8d794460d6957f833cc54525e94c36afdb5fee1ac7be06a696c31c12dc9c0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.06640625, 3.20703125, -0.986328125, -0.8125], "student_probs": [0.0049374341033399105, 0.9632214307785034, 0.014540297910571098, 0.017300788313150406], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "46ec787c572fe1c1a42a2d9845f6509bd371f6af604e53add34c9d2bd3e1dfc4:action", "state_id": "010f90c214fb0d5c5a2b60f9b4fb820b59c83f4fa008de4083d0b3c4c2f96b0d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.75, 3.203125, -0.8173828125, -0.682373046875], "student_probs": [0.006753724068403244, 0.9564409852027893, 0.017162233591079712, 0.01964300125837326], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "59e2b3f34fa86c9ebbd974b68f4e39617e728296fd8a423fae2e396fb4eb0581:action", "state_id": "7b95bb411c1f398a11aabde9e10d0b0cf4c55bf37462116256050a6d0e40948c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.69140625, 3.2109375, -0.791015625, -0.6419677734375], "student_probs": [0.007096149493008852, 0.9551764726638794, 0.017460530623793602, 0.020266935229301453], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e5ff256b379ee976856c790f06ff51405cbeb49d84d536e82047bef562238db1:action", "state_id": "be2d4e2024e1a4d524bfef2196c76706a152dde154291e42334f8d8433678544", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78515625, 3.1953125, -0.7607421875, -0.7138671875], "student_probs": [0.006568265613168478, 0.9559623599052429, 0.018295658752322197, 0.01917368732392788], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7fa91d53a51375227149ec53bf1f13cfe5450feca32b509cbdef45d501e43e8f:action", "state_id": "739f16374d26827ecdbe55a60eb4830498543d21283c7a2bbeb7b88540bfeea8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.73828125, 3.216796875, -0.69287109375, -0.5625], "student_probs": [0.00671235891059041, 0.9524413347244263, 0.019093740731477737, 0.02175256423652172], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f5ee53b75b55eaff7083b113f9cd247964eb7995b4d338c1e2efc869d8d86735:action", "state_id": "e7bdac24a5b624aeaa4a4adedd22e269c396e7fff5ccbbe8c9d7cfadfe96b88e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.875, 3.20703125, -0.71142578125, -0.6953125], "student_probs": [0.005932757630944252, 0.9557729363441467, 0.018992863595485687, 0.019301380962133408], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "225932275a2d0119d7fda4e11891efb06f5e5d9b0a97567890f51f4b3bfc8ebf:action", "state_id": "bf9c8c7f6776a30fcdb38ada6e3adaac954cb849f7dbbf154d936e47e235d9a4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8515625, 3.171875, -0.643310546875, -0.52197265625], "student_probs": [0.006247664801776409, 0.9492244720458984, 0.02091485634446144, 0.023613005876541138], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6d1702b10168117cc4c3c66c247ab541b7faca85307389e04e9ec229bdc82a9c:action", "state_id": "ffbef194d9314108cebdc9b96cc524a0303384cfd3a60ea41677204f1a06a9a8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.94921875, 1.484375, -0.3623046875, -0.228515625], "student_probs": [0.023548858240246773, 0.7297274470329285, 0.11512188613414764, 0.13160176575183868], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "51ea28455bd855625336c4951ed80d4e1cb018189e885344a0f2d83a8e4025d3:action", "state_id": "88fe45317bee53ffae5d550a468e0b8a9036dc6aa695133a2ef63adde6e401a1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.83203125, 0.86328125, 0.0390625, 0.203125], "student_probs": [0.0333789587020874, 0.4943472445011139, 0.21680958569049835, 0.2554641366004944], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c736492f96ee7f718232ce68bc54e18745b26b0aa7549093b164a02262f32644:action", "state_id": "6fd7b6c9cbb34f1126b9700e11b23f73e9607ef450a0a6da85cfd0bf8ad5a471", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.30078125, 0.65625, 0.1796875, 0.46875], "student_probs": [0.05452150106430054, 0.3859185576438904, 0.23962228000164032, 0.31993770599365234], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81598a1ced26648f210e97ebbf87ebf18ca5011c443a2a63428a9473690f8fc6:action", "state_id": "a1e134f892e4c749ad5a5b305ddfb0e141d54fa9e6cdf29c286be5bb4a867b0c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.828125, -4.2421875, -1.015625, -0.921875], "student_probs": [0.1718800812959671, 0.015374873764812946, 0.38733774423599243, 0.4254072904586792], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ca069864b4c1bac97c58ffad8c816ee4a07008d0402a652c010e0e18f9619379:action", "state_id": "b8985727bb0e672a498a2a6d0166803189ed9e709c476aefc0b7b8c9ba4c42ff", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.892578125, -3.20703125, -0.7685546875, -0.540771484375], "student_probs": [0.2737853229045868, 0.027055524289608, 0.30993661284446716, 0.3892224431037903], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2246ffae5961714879f4c6f1c3704b117d1a39afe3b256e2a16649288c0dc399:action", "state_id": "faf1dd1629e818cd48fb277064a32cbde894b94c0cfd28975be195889335e25a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.47265625, 2.150390625, -2.390625, -1.55859375], "student_probs": [0.02514553815126419, 0.9417382478713989, 0.010041352361440659, 0.023074844852089882], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "efa848aa0eed381a97288cfe6cfa3ba808e1d91a1e0e9ff03dc22fb2ca58167a:action", "state_id": "5dfd35e480c7c1cd47af3960bafe20de31e482fe61bcbab55193002b0a920ce9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.03515625, 1.015625, -1.36328125, 0.578125], "student_probs": [0.1674666553735733, 0.47893527150154114, 0.044374242424964905, 0.30922386050224304], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "80d530b4c154c8c3136aaa7ae48b171a5ea61611c738efe879f8340e294eefe2:action", "state_id": "b6aacaea48f38320caa6c036b7d6d49e27bd2aa15d17acd27e191d151c8fce6f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7099609375, -3.75, -1.33984375, -0.7861328125], "student_probs": [0.3988601863384247, 0.019078688696026802, 0.21245455741882324, 0.36960655450820923], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ecf142d3f0286903ada847c3ee1997cd8fcbb4b64abab5b6307cc4422380a83e:action", "state_id": "a9357960d1ad5a8a48d58ce90ce2cffaf19ded817c08962f49180496643e918c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -3.2265625, -1.4765625, -1.3828125], "student_probs": [0.40535858273506165, 0.045479971915483475, 0.2617191672325134, 0.28744229674339294], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d8b98b48f77537fc84bc0ec2038af3459252fa77f5d7106ba4c95a86f04ef4b8:action", "state_id": "612f8208cbb206cae3f548e1a0eab8be3e2998d25fc280ee1f793af3fa1a6238", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.11328125, -2.3125, 0.8828125, -1.4921875], "student_probs": [0.10699433833360672, 0.0322512611746788, 0.7875049710273743, 0.07324937731027603], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dc954a8930aba59ff3902607cf6c06d84954d9d30611f77390e07329896f9751:action", "state_id": "a9ff38ae1bb3429d5d00a76aff4b73eba1a7c05eb7141cda27b08420ce3d9005", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.46484375, -0.228515625, 1.2890625, 0.64453125], "student_probs": [0.2009308785200119, 0.10044411569833755, 0.4581422209739685, 0.24048276245594025], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3172c134220334b7ad51b3af964778bc99919c8a9e11a95d623c198462e7e716:action", "state_id": "72653955efa17db2dada8ebb407205aaa0304891ac1d340d45e5665cac697ee5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.458984375, -0.138671875, 2.0078125, 1.318359375], "student_probs": [0.2629912197589874, 0.053221601992845535, 0.45529645681381226, 0.22849072515964508], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ff0ae4e95a1893181fcb130c04d95bac1c32d2cfe3fcd91891d61b170c64b409:action", "state_id": "781c14c9926b19cd04312310cbcb6bcef8c8e0ab121354aa10514ca67ec489f0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.828125, 0.30078125, 1.7890625, 0.90625], "student_probs": [0.18919417262077332, 0.11165682226419449, 0.4945812225341797, 0.20456768572330475], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4bb892f2a0b54fb9872a5675708d40dd6854cf021a05a16b569a46d5e9045cff:action", "state_id": "6600a225a69d3f78c0ca3849787c3c7e59ac45e613b0c38a165a93bc05e070f4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.078125, -1.7578125, 0.23046875, -0.28515625], "student_probs": [0.2975361943244934, 0.055470336228609085, 0.4050982594490051, 0.24189521372318268], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5db1ad6d12c5a2ba69e63a75f8a535c256b2be93ee42739ce874fb20d8b900f0:action", "state_id": "47f0e3901b88f491b04665527a01f79a7f767ab39795deb9d08c8f470b34f69b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.388671875, -1.5390625, -0.91796875, -0.333984375], "student_probs": [0.33763906359672546, 0.10686718672513962, 0.1988758146762848, 0.35661792755126953], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "22ef499648235ea7a1649f79df26f85fb584b4991f09df2115ce1ff0ad743a3f:action", "state_id": "0b3fe85339c8da559659f6d43961941826a588d65932521f54aff14f45eab304", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.05859375, 0.1953125, 1.6796875, -0.515625], "student_probs": [0.12872879207134247, 0.14758828282356262, 0.6511900424957275, 0.07249292731285095], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2e4f44e91b823c2332570652e62e404b3239252a1176c545d3326c1411734351:action", "state_id": "76e1df98d21271028e6ae7c933b85098a4f2d47d9fcc54fd5d695c026bd7d0b0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.4765625, 0.1328125, 0.234375, 1.083984375], "student_probs": [0.44945213198661804, 0.1172465980052948, 0.1297801434993744, 0.30352112650871277], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c659fbc597fc416faa091672e0544ea5ca01e049258f4193ac882617b3723478:action", "state_id": "a5bb455ec2f13a59d143a21c6948cbf7a31be38eeadd3fdd7cd7996327af875d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.2578125, 0.25390625, 0.1796875, 1.2890625], "student_probs": [0.365173876285553, 0.13381622731685638, 0.12424415349960327, 0.37676575779914856], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "730f1f56721a7d5d3f0f15e317e8ab7710bc60c9e8c7d28f560aa7db75fd59b4:action", "state_id": "09b9393af170302c4548004db658f04280953f0155b27f711b1239f135cb1fa0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.107421875, -0.431640625, 0.2890625, 1.298828125], "student_probs": [0.3488345742225647, 0.07485368847846985, 0.15389007329940796, 0.4224216938018799], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "48a68623821ff079991ac8d21dd3347589ad94178bbbdb8de6df3b0ccb513609:action", "state_id": "e6af37b6c19a1c6dce067fa7f424d541fdba67b6e10fd5f532895a341d00bbe9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0, -3.32421875, -0.5848388671875, -0.15625], "student_probs": [0.4084063768386841, 0.014702887274324894, 0.22756226360797882, 0.34932848811149597], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8699f153cdf36b851836eccaa0af3afacb76c9279f7f7614079d624ba520b8f7:action", "state_id": "b75e91fc798b875156a841541a98fd85121ad8e569237db4b201eb0b2be56439", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.65625, -2.6953125, -0.85546875, -0.8212890625], "student_probs": [0.16990071535110474, 0.06010853126645088, 0.378416508436203, 0.39157426357269287], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "df0e2aa01365270f5f8297f1da3ca83db135c600f78f841f6fc79e194cb4a8f0:action", "state_id": "623d2c43c250ca60e3236a9dc4eca70b08f86df4949ad75bc9217797c48b50c6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.7109375, -0.9228515625, 1.5078125, 0.0546875], "student_probs": [0.2542860805988312, 0.04963374137878418, 0.5641583204269409, 0.13192187249660492], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "76cb1dbcea988de0ac9220733d5a7a83cbb02041bbc21da41ba76c852c9ba90d:action", "state_id": "2b4f1ca0739f5474236e992696090bb2c93ee9d06cbb9f7147bfe50aec0e9873", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.5, 0.03515625, -0.00390625, 1.23046875], "student_probs": [0.4510372281074524, 0.10424106568098068, 0.10024765133857727, 0.34447401762008667], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0a3c413848b8867441b07ce23347295ae4744609ad5a75fd6caf4cf78405e504:action", "state_id": "39b6d6134d60a0ac1ba14ef266dd4d762a6742b2d04f34180ca49b85d0aee848", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.869140625, -0.28515625, -0.0546875, 1.150390625], "student_probs": [0.3292657732963562, 0.1038106307387352, 0.13071732223033905, 0.43620628118515015], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ac0b47516e376e8fa6dea67bbb8db8c81cc3c8cc2167ff81ac9427600886e1cd:action", "state_id": "9ccd5f7d0fd730583f31dec92790d1a779b8ad6faf08225921082a5dd257f201", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.1328125, -2.94921875, -0.353515625, -0.01171875], "student_probs": [0.39585554599761963, 0.01815630868077278, 0.24340367317199707, 0.34258440136909485], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2082d5f69010d99e1b5ea1e89b3738130a995e3c21e5f69427a7349bc4dfc38e:action", "state_id": "419807054b3a07e6b3aef8859a83cf5704ddca1b50c34da84e13300171f51916", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.03515625, -2.8203125, -0.45361328125, -0.15625], "student_probs": [0.4005276560783386, 0.023041894659399986, 0.2456759214401245, 0.3307545483112335], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5de51d08adb2b5ece675f2d33cc05796e6edbdbc119a98f05d8126933aff1cee:action", "state_id": "069d9c64165510cb841c69e6c55c6f57504cfba1465340773605ecd1bde343b9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4140625, 2.09375, -2.609375, -1.45703125], "student_probs": [0.028061719611287117, 0.9365649223327637, 0.008491738699376583, 0.02688148058950901], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "df2267e3b51006e143d8a18a905ff20d2c49e0aebeafcf30765509019f9e341c:action", "state_id": "7de9f44f4c485deb1b64b5384a651393268669b05ef74254d78d26fbe35422a6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1328125, 0.85546875, -2.09765625, -0.2265625], "student_probs": [0.08961346000432968, 0.6544445157051086, 0.03414655849337578, 0.2217954397201538], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "beeaec5fddc216f17496285f16eb45556348090a8cf54dc604f4ac53bb1b83ab:action", "state_id": "d8a4bb7b6af0c359b12432b02ac0b16773e18423b6dbfb53e7bf4dedce160174", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.30078125, 1.4296875, -1.91796875, -0.431640625], "student_probs": [0.05190923064947128, 0.7962915301322937, 0.02800293080508709, 0.12379627674818039], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e45f484ab0e94403cd514ee9e8057691aff674377144c08bc90ae68cc4ad3abc:action", "state_id": "09bb50e688dc8b52a74fa61c84466a7ea5681266efe626d06aef11ec1b804314", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.44140625, 1.955078125, -1.75, -0.396484375], "student_probs": [0.029038872569799423, 0.8670700788497925, 0.02132844552397728, 0.0825626477599144], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "69b2fee1b8882e7074cf37f4037e3dfe9f4ed6a8712d65ed590397e3ab5cd4c2:action", "state_id": "7476408d3338c87bf862a87a61068bedf973b8cc55abdbf85e77d445e7ea265d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.96875, 2.984375, -1.890625, -1.2890625], "student_probs": [0.006864767521619797, 0.9721665978431702, 0.007422583643347025, 0.013545977883040905], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5a2bd93178e1323e5fea5d9089b1484e4b7be76fca6a9ef194d6b071c174934a:action", "state_id": "5209d64ef2c786a24c1f5fe74c14b6c08c58d28172ac6f3f40d1fbaa8f3a551e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.00390625, 3.0, -1.734375, -1.453125], "student_probs": [0.006534324958920479, 0.9735754132270813, 0.008555721491575241, 0.011334490031003952], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "13675056e09c44e28d1b2a43a2eaf9ab8967822038f28b094564ee46c31e56d6:action", "state_id": "468bf24cdaefdc045da5cb20480b1ba0272e9ed1cfe46d1f6f84695b24e74a6f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.4296875, 2.96484375, -1.8203125, -1.59375], "student_probs": [0.0044376361183822155, 0.9771626591682434, 0.008162062615156174, 0.010237520560622215], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b99775232c29200bf464cec0f9e8b065465e2fa824d540ca1bd56730ff1e7491:action", "state_id": "1b3f078ae61888079960547b0767ac7c8ace2c781634551114e002ae47051cc3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.33203125, 2.66796875, -1.205078125, -0.544921875], "student_probs": [0.01696912758052349, 0.9264829158782959, 0.01926613226532936, 0.037281788885593414], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "626441619b341a73f854649543ba61891e7fff594cbd2f7c2b11ae8b0960e9f5:action", "state_id": "1416813b0255f143b15eb1ea5a2472acc963a564086e0c289317a36eb4302eb3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6484375, 2.65234375, -1.66796875, -1.1875], "student_probs": [0.012932660058140755, 0.9538792967796326, 0.012682519853115082, 0.020505506545305252], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "eb1b7752404ff43801cb5d7cdcbc46d34f2af8015eb22ef8ca4e00d9b42dadc8:action", "state_id": "fa910c24d6402e58792d014133c816f08e57c6093124930aa37b46ad122ca8f2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0703125, 2.822265625, -1.85546875, -1.3359375], "student_probs": [0.007266351953148842, 0.9685813784599304, 0.009007865563035011, 0.015144377946853638], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c95a1ad5f12fb6d21b674f37b3bc4d971b90d4501c79d55ecb75d94ed7eca7d2:action", "state_id": "0a61176d32684ef887496730f6ef9f6853e51ef37bf2bdebc8a9258ef579e9c1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.97265625, 3.078125, -1.83203125, -1.625], "student_probs": [0.00626130448654294, 0.9776676297187805, 0.007206717040389776, 0.008864413015544415], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "92118bb5d1cc33f3130e99b458b529d602c7660b84243aaca242e4b58779d7cf:action", "state_id": "0b6e721c008deec566a8e0225cb91223fe1ac003754836752203b77756d99d96", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3515625, 3.259765625, -1.31640625, -1.27734375], "student_probs": [0.009640364907681942, 0.9699913263320923, 0.009985312819480896, 0.010383082553744316], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6bc64bee40214eb313ac2df861f128cf6788ab1f96d1ffb241ab05e2adfcde43:action", "state_id": "3b1d2130557aeeb8dbe4ff2527dd164d16fa90f356d615782671b227202a61a1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.59814453125, 3.01171875, -2.32421875, -0.4814453125], "student_probs": [0.025469426065683365, 0.9413754343986511, 0.004533093422651291, 0.028622066602110863], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4af4bfac5a820ba548046b452d723db3186d65fe60fc8323e68244fb920052cd:action", "state_id": "364b845ba6151580b89c150888f749ae93bb5fb4eebceeca77a73cdaa96325c1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.40234375, 2.83984375, -1.828125, -0.447265625], "student_probs": [0.03598930314183235, 0.9209532737731934, 0.00864897295832634, 0.03440837189555168], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5800ff8635c1e505921c60e30b5cab496fd22baea3285bbcc5a2bb68bfb9ec28:action", "state_id": "0f0b2e070baea2fc2d544db448451f80903d1587bb3a177adce9bbf3700db79b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.66796875, 2.7890625, -2.2578125, -1.109375], "student_probs": [0.011168970726430416, 0.9631129503250122, 0.006192232482135296, 0.019525732845067978], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cc6f6da2bb66785cbefb021d3a0e40212b50271aa7cb9bac81e2fd7acb13d785:action", "state_id": "223e865992555ecf3839475832b78a9e5ac9e073b72a8c7a8ac34cd489a0e4f1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.69140625, 2.8828125, -1.91015625, -1.056640625], "student_probs": [0.00993617344647646, 0.9633345007896423, 0.007983939722180367, 0.018745383247733116], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e4fcc4b8f864a1af6466da9bff0a15585c1022e3b04f448f95e26b0f32819d31:action", "state_id": "160bd77e8a2116ff44ada0d60ae0351260b54066b683d23cb34d5b891213723b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.35546875, 2.99609375, -2.28515625, -1.71875], "student_probs": [0.004653310868889093, 0.9815583825111389, 0.004992273636162281, 0.008796005509793758], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0a4103cfc540c29451f5c0bc233f509b519418824378c8f4ea9daf3722e2ef40:action", "state_id": "4b4bdf852f2b93efcff1f2122ba491ee635d55aa2c6a21ecf8e174fee4be1375", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.84375, 3.041015625, -2.1015625, -1.00390625], "student_probs": [0.007334163412451744, 0.9700124859809875, 0.005667402409017086, 0.01698596030473709], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "744d9d609c6dacc90c7249da654d5d895a7153f24c9dce66164eddde92e6c95f:action", "state_id": "5caeeb0627ca8b85aa5e0c5d9cc6e0b1f768f15e3f28a62ab0c6115228585dd5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6409912109375, 2.943359375, -1.671875, -0.095703125], "student_probs": [0.025567766278982162, 0.9212053418159485, 0.009119806811213493, 0.04410708695650101], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7b3a323f054a44ec83b3c6696b0bcb68115a1099ac021df39a19d56171cebb0b:action", "state_id": "7e52adf95ac69777e40513e968d19ab9f7a9fcfd63c6f02d132f837cf0f1fa1e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5791015625, 2.904296875, -1.69921875, -0.126953125], "student_probs": [0.028194306418299675, 0.9182948470115662, 0.009198155254125595, 0.044312573969364166], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "08e04a22bcdefcd441622f6549bc50f3a3455865cd06cd41b517848cacf3abae:action", "state_id": "aee50093bcb1fe78954f6114580bc22c32931d6d046af106719bf88e31abf543", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.26171875, 3.046875, -1.6015625, 0.05859375], "student_probs": [0.03334880247712135, 0.9119777083396912, 0.008733603172004223, 0.045939911156892776], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c075c1200b88d786ea6fa036562e2f33df5c337b261c121e5e986ad61947b057:action", "state_id": "5d69d56d345a5932087663c6b9cbbc70bd5289861ccb49a6e60dd9ec648498ca", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.03515625, 3.041015625, -1.8515625, 0.1640625], "student_probs": [0.041565652936697006, 0.9009466767311096, 0.006758952513337135, 0.05072875693440437], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c2f5a2db73f8312092e787199f700bd9badcacdc5f8259ea0cbd86adb78c898a:action", "state_id": "6e2b526c2eb5e45a5046fc42db7b0e053b3963872b74358cfc25f05c30f2758c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.220703125, 3.162109375, -1.9375, 0.0078125], "student_probs": [0.03135792165994644, 0.9236003756523132, 0.005633157677948475, 0.03940854221582413], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e339fe5578f75377988284bd13d1e376b137fc905406cf44a3c12bb3af362f4b:action", "state_id": "052bbc41f0a649b6fc27e79412a3ccbfa6b9ed6f06a16d7625592094f467997e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.578125, 2.87890625, -0.83837890625, 1.0], "student_probs": [0.07843533903360367, 0.7829397916793823, 0.019025318324565887, 0.1195996031165123], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "73c8d9ed12c0deb7046e1d38bcd9b9446e66cad263172ce98bf140208a4b4033:action", "state_id": "afbc1b549714fea678d3d33b485adb8783119ced21d510723848af506f909ceb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.33984375, 2.8828125, -1.01953125, 0.890625], "student_probs": [0.06365858763456345, 0.8095698356628418, 0.016348877921700478, 0.11042268574237823], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a26f8928caac71bd340dd3eb55f27120589b8dba0c47001f88560c67423d33d2:action", "state_id": "7fb5d0ac32655bc5958a93df59b0a08342902c1846db5e41c5cb102eb9cf88f6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.4375, 3.046875, -0.9306640625, 0.810546875], "student_probs": [0.06135992705821991, 0.8339154124259949, 0.015620636753737926, 0.08910396695137024], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "14281a1c1e8fcae1279047a73c844d8ab2e426dc26637065e64eb58ce4747616:action", "state_id": "8f5d78c31c036c849d5ecb6289c52865d9626fcc6ffad51b32217b5ce62a88b7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.125, 3.0625, -1.173828125, 0.3359375], "student_probs": [0.046780776232481, 0.8826884627342224, 0.012764197774231434, 0.057766545563936234], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8467dcf46e558d23fb3f4046bfcabc3222b067cff112cbb5f1df8792697ddc1d:action", "state_id": "f88ba6813f2aa68d848dd9205c08f53ae5831cda67c0ae62c9f228add803ae00", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.53515625, 2.70703125, -0.541015625, 0.93359375], "student_probs": [0.08616890758275986, 0.7561081051826477, 0.029374809935688972, 0.1283482015132904], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fa07af7464aba4533ba2e6931d5c762cb13cdf39ee9f8bab9eb8ab8b3fb6b6b5:action", "state_id": "f29f60c5cb5b6102bd28c6d632731b7d08776ca7996e9c849780644566463c8f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.404296875, 2.75, -1.3125, 0.30078125], "student_probs": [0.03722481057047844, 0.8724212050437927, 0.015010835602879524, 0.07534319162368774], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e08a0fb30e53c30270086979f81a9827d78a8e2d1d629f69125353264989ddea:action", "state_id": "72e8a2915faeab6b80ec384058e698c52c6e71a8fc293cae92eb64806afbe815", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.12890625, 2.822265625, -0.8759765625, 0.66796875], "student_probs": [0.0438198558986187, 0.8382018208503723, 0.020759765058755875, 0.0972186028957367], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0d27c132f40c071a3c2336a1dfea2f34f642e4195c28d260f7e0837570f1731a:action", "state_id": "055171399fc49751ec00f0161b780ee0d11dd5373e2f379ac64e0cb5acfe32d3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.07421875, 2.865234375, -1.083984375, 0.6484375], "student_probs": [0.044783394783735275, 0.8466526865959167, 0.016314785927534103, 0.09224920719861984], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6e29532681ab65ec1deb00a99cd677c12846414c1fa5c0a5918d886d3e0794c2:action", "state_id": "f163fd402136ce5d9e191b51897c83fc0ea598a286358379265e3f38502abdb9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.48095703125, -3.48046875, -1.060546875, -0.49609375], "student_probs": [0.38536885380744934, 0.019195755943655968, 0.2158558964729309, 0.3795795440673828], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0b92349179d0765dc4601aad2e97a2711eb15922f3ce82a9d846ee1de2085daf:action", "state_id": "b66ef0b68269bdd7d8773f78049ea25a206d0446bf46ee64b7df3ae2fae71063", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.548095703125, -3.30078125, -0.59234619140625, -0.7265625], "student_probs": [0.35001474618911743, 0.02231568470597267, 0.3348641097545624, 0.2928054928779602], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1b67330cc393566841bb9418f8ad009d4784441d7c7841fb7f955b461a83e73a:action", "state_id": "21a1e5b1e92da444fe8b5f37edbf6ce099395f5cca82e507910530bdf1efa941", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3671875, 2.244140625, -0.78125, -1.001953125], "student_probs": [0.024240834638476372, 0.8972788453102112, 0.04355289041996002, 0.03492744266986847], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "93978794633dcb1392f276e16d6b3e9b784d5852fb3f7aa670e9a195ef752999:action", "state_id": "7903bc3339eb3ab2fdc6ad3339ccc152ddd6716e6995ea7240fb531d77d5841e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.15234375, 1.140625, -0.7843017578125, 0.8203125], "student_probs": [0.165869802236557, 0.44562798738479614, 0.06501107662916183, 0.3234912157058716], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5a480b728d284957e1b5369a84f0ef257fb93cd95a33110e8fd21dac82e0bcf9:action", "state_id": "5c8246fe7dc90ad92b985a3babc77f3065b8ac3c35b1cb172e768a9616985a7a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.598388671875, -2.1953125, -0.6746826171875, -0.91015625], "student_probs": [0.34950196743011475, 0.07078062742948532, 0.32382890582084656, 0.2558884620666504], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7670e2f9d27969cb6424c538eadb7e8c0c8931d95c615e4a2fee888602f14a26:action", "state_id": "034e50f1750ad717a4be7cd655c03d027634e21d2d969bae3d4f9764b8ae419a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4384765625, -1.99609375, -0.32421875, -0.64910888671875], "student_probs": [0.31829389929771423, 0.06704459339380264, 0.3568205237388611, 0.25784093141555786], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ba46f5e2da1b5f2ab3083cc72cad95f91c9c7410059a9f76fddd409d0ee93593:action", "state_id": "28e8ccc86297f931155fd60e6511bd640a230d8fabf88c5ab1322c6b28766b10", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.44921875, 2.29296875, -1.45703125, -1.234375], "student_probs": [0.022015733644366264, 0.9288477301597595, 0.02184440568089485, 0.027292203158140182], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "24ace43cbecd0be997ee131957bb1981705dcb3c62e2129f9ca390af906b9df3:action", "state_id": "61e128667991480ead31a2dc28498a2d6381c55bc2f284eee3d5f53c61f23542", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.517578125, 0.9140625, -1.310546875, -0.02734375], "student_probs": [0.13753724098205566, 0.5756704211235046, 0.0622355192899704, 0.22455689311027527], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e43575037e6705ed25d0c0d368e248ad4baf7639dbf3dc3d1adeb243cd47dcf2:action", "state_id": "b63d90dc04c2404c15784ef51f124d28a263c91595bcedf032cd27f1ffd8e658", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.90234375, 1.126953125, -0.43359375, 0.1484375], "student_probs": [0.07653091847896576, 0.5823034644126892, 0.12229606509208679, 0.21886959671974182], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "37fcdf796c26adb889394821a5b9c5c55706d2019fd982736839669e27ab1774:action", "state_id": "175b3898e6b7e9cff00d76d17a6f6250020da90ebceb68149c82bfa1db71fb63", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.68408203125, 1.853515625, -0.67333984375, 0.08984375], "student_probs": [0.05942367762327194, 0.7516647577285767, 0.060065459460020065, 0.12884609401226044], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e2fc9292efff4572af0b858b31772abc4d0c63fd51d3622ef16d531e5d23356d:action", "state_id": "2416ceab6150b669d861968737e9d357a7aa320615f0550ed7dd53795867bbbf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.11328125, 2.830078125, -1.6484375, -1.41015625], "student_probs": [0.006903578992933035, 0.968161940574646, 0.010988879017531872, 0.01394561305642128], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e582fe073a0abac531cc06db58c0a3625f339b2df6cabbb72379b8fb5bb9dd74:action", "state_id": "462b3bfa7d5476a807d78efe393794c9584dc1cb91d249cd26283cb1c6f859b4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.23046875, 2.880859375, -1.796875, -1.51171875], "student_probs": [0.0058656116016209126, 0.9730494618415833, 0.009049419313669205, 0.012035454623401165], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4533fba5d66b0ef203cf63a86b690c02ba28e0603ec74c2259582881962df7fe:action", "state_id": "26657483d8b8841d6f73e8e63fc428dd2e989ebd40ad49a3a24f37e1f115c952", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9921875, 2.859375, -1.7578125, -1.29296875], "student_probs": [0.007563355844467878, 0.9676567912101746, 0.009560978040099144, 0.015218834392726421], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d7b4b36f37e872be3c2552978255aebfd8310b11557d310f11cb634c9cd3751f:action", "state_id": "b7e00f08d8e5f83f57bddc787b006fe349e90c8b41cc1f626ab33740906cc953", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.44921875, 2.75, -2.29296875, -0.7314453125], "student_probs": [0.014262443408370018, 0.950367271900177, 0.006134200841188431, 0.029236068949103355], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "72d68abde557e0197a3b73bbd165cda69228bf73aa85390fc704ee273560f092:action", "state_id": "92231a9718bbbc0583710c9b00b7cf3107f043bc13f590387b6917f69113be5d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.82421875, 2.640625, -2.7109375, -1.203125], "student_probs": [0.011088883504271507, 0.9637064933776855, 0.004568679723888636, 0.020635992288589478], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fff2b435aa7e5259510c1473586e34328a8de400efabb27b3271e4e1be727e0f:action", "state_id": "4db787ca3910360005175281fe20eb1e479a456da1a09c90d992e82e7d115f0d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.36328125, 2.990234375, -2.3203125, -0.90625], "student_probs": [0.012389312498271465, 0.963285505771637, 0.004757883492857218, 0.01956741139292717], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d1d2e3ad4c15ecae81914293f4e9d8154d715921f28c7595ca38e5abd30ce967:action", "state_id": "e302d12053d556a6ddba1c6a5bd7d7499e1655c7d9ed0b2b429b00a0e045962b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.28515625, 3.173828125, -1.95703125, -1.087890625], "student_probs": [0.011219751089811325, 0.9693832993507385, 0.005730488337576389, 0.01366641465574503], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "69520d9b651f4e6255123203a10822be03ceb23c5139f43fc7d5c6e4d62838a0:action", "state_id": "245e92042af38274bdc8323e4f2ff9051b46e738d4a5b2dd6f7ef9dfb394bc20", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.470703125, 3.3447265625, -0.9052734375, -0.76806640625], "student_probs": [0.020926378667354584, 0.949979305267334, 0.01355072669684887, 0.01554357260465622], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ae9b24b035ff5831abf9689e724ae5787e9d1b234655ff5d36a3d1a5c6279081:action", "state_id": "fb26a11a1f783a65c856184fb220764361748467a1e6a2dd22b930cb1b3fade2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6171875, 3.46484375, -1.005859375, -0.8193359375], "student_probs": [0.01619153656065464, 0.9596032500267029, 0.010977160185575485, 0.013228058815002441], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5938b6a837b91be0034a5a2d7e8883b4c3e0ce6765ba1771533c8a2d1735f034:action", "state_id": "553b0f627d9125fc9036eb8aa631f35a4d2d2bc74076893aff5d2ed74ad6a042", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7034912109375, 3.4736328125, -1.0390625, -0.79345703125], "student_probs": [0.014747734181582928, 0.9612298011779785, 0.010543590411543846, 0.013478875160217285], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "00b3628b4815b96cb39de5339cd7d41effb08bf15803b1c9a2b6925c91173800:action", "state_id": "087c0987946babd77ee563aaf533c4d2338e5c61f67f92841878738482823229", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.699462890625, 3.595703125, -1.162109375, -0.74462890625], "student_probs": [0.013170052319765091, 0.9659494161605835, 0.008292064070701599, 0.012588446028530598], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1955e40436bd80a5e2c15bba3831f4edd5d2a2613fa3acc920ebd96225617b8f:action", "state_id": "afc3e06fb4ef642bbe7cb700b6d02299355f96d58122101a057f32bc691f2598", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.966796875, 3.443359375, -1.5546875, -1.056640625], "student_probs": [0.011799146421253681, 0.9708611369132996, 0.006554400082677603, 0.010785292834043503], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e24c7bbf82aafa00bb5e3a98426964f53582a784bb110422ae31affe084bc1eb:action", "state_id": "2d3dbbe6dfe4fd63cb0798597a3f9b72b800e8fedcc7fe6b28b3eae9e9221ce2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.759521484375, 3.60546875, -1.38671875, -1.0234375], "student_probs": [0.01235319022089243, 0.9715615510940552, 0.006597673986107111, 0.00948772020637989], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e7008d6a44bb1ecccc4f25a16097c3f634af0e7f1716955a6590f115edfe780f:action", "state_id": "ff57f665fdf94965af9df0faefe0998f6ad57fa050cf6e6f5f3b2a6d9ea0375e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7335205078125, 3.5859375, -1.30078125, -0.8427734375], "student_probs": [0.012884693220257759, 0.968257486820221, 0.0073066093027591705, 0.011551174335181713], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "63946686b2ff9505c00c0892223734d0b0cb0cf07f7e956f2d61ed468dafb89a:action", "state_id": "707d4e89a46dfd58102e02d65a23445d3ebcaf728c436f04667dba12e5870644", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6734619140625, 3.546875, -1.23828125, -0.765625], "student_probs": [0.014176989905536175, 0.9648350477218628, 0.008059092797338963, 0.012928796000778675], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bd2ef162342629d31af6f3803eaa57bca0f6f47f46333fa0b15c4cde9f90b2d8:action", "state_id": "365a41c105087b33b2e282b8d245735e1175184128b462b3c8af1705281212e7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5478515625, 3.546875, -1.166015625, -0.640625], "student_probs": [0.016006847843527794, 0.9607778191566467, 0.008626618422567844, 0.014588640071451664], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "601e20c66229fea0be08572770901bf8710e40b10992498d84df2e4b30f5f6f9:action", "state_id": "efaa77126218c032a9bf11503fd64b033c98cb7342cfaeff171a3decfc2d4528", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.439453125, 3.63671875, -0.861328125, -0.5126953125], "student_probs": [0.016258925199508667, 0.9579675793647766, 0.010662863962352276, 0.015110651031136513], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9b177bce82084c929025a6d9594dd8825dd64917d2e4a09b9c82178268bdda3e:action", "state_id": "632728564beac59d91b6f47e1a62d1b37109963d0886fb6e105a42f6d6b0d8e2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5048828125, 3.7392578125, -0.9140625, -0.5908203125], "student_probs": [0.013835528865456581, 0.9642787575721741, 0.009189487434923649, 0.01269619446247816], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bdf5e316f100e9a76c2eae130def13682b13792a22bd2a188739a6c83dde5a4c:action", "state_id": "7a5e83752102630a9cb8d43f0dd4e3a8c6bd6fbc94ec9650f8bb608c88fdeb84", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5078125, 3.1640625, -1.5078125, -0.109375], "student_probs": [0.023706261068582535, 0.9322623014450073, 0.008721047081053257, 0.03531036898493767], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4108e1a9935a028ba36ccf18862da1d6b17d3c4827054bdd0644361dee50377a:action", "state_id": "65164d20d08f8af07006edb35e5c32cf0694b1a18871e3e115e629351b6eadf3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.17578125, 3.1875, -1.103515625, -0.10546875], "student_probs": [0.03189578279852867, 0.921271800994873, 0.0126131447032094, 0.03421918302774429], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "70f059e66f4283d3a4647675c6deea242a4fce439ae5ed9d03b962d3e0792c19:action", "state_id": "723110a828118e126fe40041a9588a2f5d4a027f9d5a696ca842f1ccb7dd4a53", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37109375, 3.0, -1.3984375, -0.69384765625], "student_probs": [0.012037806212902069, 0.9525532722473145, 0.011713107116520405, 0.0236958134919405], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "547eb5fbc53840b5db326beea6d74640ac991235fd136f31d25a4903377be8be:action", "state_id": "be729b28b0e881886fc1edd8d25a27aa199f807978dc587e119fae0b2ac27b7f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.87890625, 2.943359375, -1.78515625, -1.1484375], "student_probs": [0.00778691191226244, 0.9674947261810303, 0.008552249521017075, 0.016166046261787415], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8f227095cd4dbc97b787acf54533492ff0c35e1fd5ebc4acbf54886120ce17ca:action", "state_id": "d12504e416de0dbde6eca1c85d67d695e8349efae06759f7591bc3b9354c2165", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.484375, 2.869140625, -1.58984375, -0.682861328125], "student_probs": [0.012212973088026047, 0.949574887752533, 0.010990485548973083, 0.027221646159887314], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "319c5020d3b2984a0b09e5cb1f7e64d60d09130fff84565d0a0001714162c276:action", "state_id": "6ca46d485756220df65a69f81e8c53063f0245371b87d32a2201d71cc7202f72", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.509765625, 2.50390625, -1.46484375, 0.1328125], "student_probs": [0.04228653386235237, 0.8610396981239319, 0.01627110131084919, 0.08040262758731842], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "69d49c920a706e353b830de2df8d264a7ce942b30f865ebdfb50c30e0b503a88:action", "state_id": "c9c840f3c704b086e918c5fd83e1416e461ed803f782323e397cfdfb9d866aa0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.248046875, 2.388671875, -1.54296875, 0.421875], "student_probs": [0.058155421167612076, 0.8122740983963013, 0.01592988893389702, 0.11364062130451202], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b75ae468f11acd5c8795a714f8f6fb731d6acfd662727bbad9b203dea7ccec72:action", "state_id": "1eee6cb9193e911184c0b6366496f79caf91837ccdade2e586b528c534e4398e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.078125, 2.5234375, -1.31640625, 0.60546875], "student_probs": [0.05968133360147476, 0.8047903180122375, 0.017300546169281006, 0.1182277724146843], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b9e64c44b4a7f0bdcb44ea6ae1c71ea915d293094d4a078523bd5040c3adb7a7:action", "state_id": "3aab338e325360cd1ca086c6bc83b199ea6eadc72cc6d6876cc6ec6c5937c6c8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.6953125, 2.345703125, -0.83935546875, 1.10546875], "student_probs": [0.1260780543088913, 0.6567423343658447, 0.027173254638910294, 0.1900063455104828], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3a7f8c77470105f0f94378436f857426dafdce13960b81fe1890bfadf9400750:action", "state_id": "1d3544df04e8f83e9366a439c62e40f32d96bc237a4b7801faf9be41cde8c3eb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.71484375, 2.458984375, -0.5908203125, 1.10546875], "student_probs": [0.11806543171405792, 0.675450325012207, 0.03199484944343567, 0.17448939383029938], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6947a58a3858c3b353ad4d91cf458aa449e90d7f4ec1ffd8e4cf3a02a0f7d7db:action", "state_id": "bba246c728659ba6ae4db2348edd7e3aa8b80ac2cd77fde2b71990e8a90ad562", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.14453125, 2.361328125, -1.078125, 0.736328125], "student_probs": [0.08143611252307892, 0.74741131067276, 0.023978618904948235, 0.14717401564121246], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "350eb9d6195947e7d6252691b205b0c17b43c55d5a1340fa933ef439b8d4a511:action", "state_id": "5fc8652fda26d19bf6d19cae4550c136a216805e83e8fccd7dc5cc637d85ce94", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.125, 2.587890625, -1.166015625, 0.55078125], "student_probs": [0.05437310412526131, 0.8195539712905884, 0.01919892057776451, 0.1068740263581276], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "51d35e2c128104cf52823d8f2a9cf1bc6ee2ea87eedc58b6f7b14a0b7ed94561:action", "state_id": "208a222a7b876ccc4c98405774a6e07a0a1695bc25f490dd8c16af59c983c1ac", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.201171875, 2.548828125, -1.375, 0.3828125], "student_probs": [0.05334760993719101, 0.8344970345497131, 0.01649407297372818, 0.09566127508878708], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a36934bcc88bf8e0e3f9b2b5d60b91678530777ec250977c01f9fc36674f0c9a:action", "state_id": "2846d203fca0241d271433a10ebd5d8b6de0804cff44285cb1c42a941f714404", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.48828125, 2.17578125, -0.6854248046875, 0.96875], "student_probs": [0.12001921236515045, 0.6488177180290222, 0.037112198770046234, 0.19405090808868408], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "96662736ea7cdcaf817665bb2d7b2fef6890a0046ee36ca56bace11dbc5284f9:action", "state_id": "144e18c16d3c1f2f66a0bd0e1b7cecb6a2416a905613672733bad39942463f16", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.69921875, -3.625, -1.11328125, -0.6826171875], "student_probs": [0.3661229610443115, 0.019632533192634583, 0.2419925034046173, 0.3722519278526306], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "294a586abe92f5338a29049410248f11ed13a327d47404ad9bac822bf4d209c1:action", "state_id": "d9773cb0a240d83fca63b38ef0cba5d834fd6ff3c5dac20e75d686c2de84c803", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.67333984375, -3.453125, -0.68701171875, -0.861328125], "student_probs": [0.3475725054740906, 0.021567512303590775, 0.3428528904914856, 0.28800708055496216], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "78c926e227408bf2c7322a50a8ec64a06753764d3d6d06d5cb0b484b255fc65a:action", "state_id": "1045605c73a743ddfa8ae7a2d6033385e42cfe1f385b0d96d2e216eedecb07ad", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.51953125, 2.484375, -1.03125, -1.38671875], "student_probs": [0.006348082330077887, 0.9458264112472534, 0.0281186792999506, 0.019706830382347107], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d557a91c0a5c071bae123aee67ea2a374c63e6d40b3ffc748e3f937968fef03e:action", "state_id": "da3b036b7091526f551795b446423a9f7fa373f7a7123bfba4fdc7507a4953c3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.76953125, 0.35546875, -0.541015625, -0.1328125], "student_probs": [0.055780742317438126, 0.46704643964767456, 0.1905556470155716, 0.2866171598434448], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7b83883802b3ec3fbe8fa3302f092104b614df273485e7a00ec878b40f39def8:action", "state_id": "cca1e38276b39a506923e75c889abb543fa340ba0c107c9786aecac143ea5ef8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.890625, 1.1171875, -1.064453125, -0.5537109375], "student_probs": [0.03658326342701912, 0.7405575513839722, 0.08357653021812439, 0.13928259909152985], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c4f385702b0ef1d54ef41810e4a6edad781eaed44f77962d0d3ec00755228557:action", "state_id": "fb4de4b1fba3b4f13eaf7d1e4f283d4325e5476f2d94d521f012b33c60b9a04a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.76171875, 1.5703125, -0.826171875, -0.37890625], "student_probs": [0.02814534865319729, 0.7879332304000854, 0.07173142582178116, 0.11219009011983871], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e980de2a680cef3870c91ea1c6281b302c388fd1edd1f50abc4d130e71935b0c:action", "state_id": "38291ad4988ded63ec3870ac32384b26f9e340d66c7cb8879a35c9c9abd1ca27", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1015625, 3.0546875, -1.390625, -1.20703125], "student_probs": [0.005586759652942419, 0.9693729281425476, 0.011374077759683132, 0.013666268438100815], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0f7de4d2ecf66a36bcb624503e4f612ea0b89ab63c85de04a435374cd5e35e37:action", "state_id": "c368c89c254328d6b3cbd0f7297eafb8bc82e3344be8bcf4a1f002d51c4c1c1b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1796875, 3.064453125, -1.48828125, -1.3359375], "student_probs": [0.00513414153829217, 0.9726781845092773, 0.010250422172248363, 0.011937237344682217], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2c23abceafe98b96a6d4662479e21c0648a80e3c59ff6b6313df2e7a97c9b216:action", "state_id": "aeb04e794d5353a4050ef5fbacef6a53c2608be923528b8545397346c424e7e4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3125, 3.052734375, -1.6015625, -1.546875], "student_probs": [0.00456563476473093, 0.9763215184211731, 0.009295172058045864, 0.009817657992243767], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "84b97e91801addadaf2e7c004f95854ce858721f7837eede12b1d88b7ff15928:action", "state_id": "d46427d1bc5a66707cc38803d4e5eda0e2b4e9027055d58409b3497fc9e9be1c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.375, 3.080078125, -1.609375, -1.58984375], "student_probs": [0.004179095849394798, 0.9776707887649536, 0.008986467495560646, 0.009163710288703442], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a095c94eaa474059d5eeaae125b92d5391e03768c5d7be2c2366438e8f0c9a43:action", "state_id": "ff3c979bc7351a16f856bcd9d5b4d9a24615e7c925e2a2374cd55e332d547803", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.359375, 3.080078125, -1.53125, -1.51953125], "student_probs": [0.0042387028224766254, 0.976241946220398, 0.009702487848699093, 0.009816857986152172], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2a9f23548c8a1a1df28f1b1cba45c5e0c19ced6c070a2b4cc43706280a38cb45:action", "state_id": "9ca511849a6f67cf78afab7a6b57542338e5d5650e40d002c17ed5dc89ed2394", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9140625, 3.125, -1.61328125, -1.4453125], "student_probs": [0.006318149622529745, 0.975050151348114, 0.008535276167094707, 0.010096374899148941], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0f5dcedd05187a6a94015d72eaa478003b83f98764999de9d5671936cc49cfdc:action", "state_id": "43c2c9474a564eea3cc77800db4f48e52bf2e9e2a97ebb13b49722d4b02b6500", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.14453125, 3.08203125, -1.83984375, -1.65625], "student_probs": [0.005259351804852486, 0.9790377616882324, 0.00713273836299777, 0.008570182137191296], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "13b2a89c2f6e9e36b8ccd4b3a3ca6d641da001b152d8b58f58984d621109f765:action", "state_id": "5f185fe807112ef4cb8a1786a751ae5d3fd197c37110fa2232deb61147ec9674", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.18359375, 3.048828125, -1.53515625, -1.3515625], "student_probs": [0.005195985082536936, 0.972926139831543, 0.009937582537531853, 0.011940279975533485], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "abbba171683d0b628b313bba1425a6f8121b49b9b7dbd253233ce6049a22401b:action", "state_id": "5e86f0e74310371711b97e31d0c4b58e32ede58cfae88372991556fac067b0da", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3671875, 3.078125, -1.69921875, -1.61328125], "student_probs": [0.004223950672894716, 0.9785611629486084, 0.008237851783633232, 0.008977102115750313], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d55dc8bea11bca05746b7abde0c8390bac81fb7ae38d45a3d55e78f0a321c112:action", "state_id": "b971993998014f560fb066810e459fec9823572e59d0a9030a3d58f02e3a8e0d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.29296875, 3.099609375, -1.61328125, -1.4765625], "student_probs": [0.004444348160177469, 0.9767310619354248, 0.008769859559834003, 0.01005469262599945], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d21a8b018b61db9ab3222768ac14a0db07a6d4c171a046650a5a2c6c0f3999d3:action", "state_id": "106ee89b334a2244ed9f7fd93d7cd4be173edbc8fb510479cd440cd08ad3dc49", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2890625, 3.197265625, -1.41796875, -1.2734375], "student_probs": [0.004040078725665808, 0.975151002407074, 0.00965386163443327, 0.011155014857649803], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c8de21bc5fb523af1206af25095606965cc72bcf5ddb610d50a43eb66f4c5ec3:action", "state_id": "8f66ee89c81bea5ab2503e6c24a928e3f3d501aa0150a2c150959111f2be5248", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.26953125, 3.15234375, -1.34375, -1.21484375], "student_probs": [0.004297416657209396, 0.9725183844566345, 0.010845988057553768, 0.012338216416537762], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "18883d338b4cb7da06b604cb0ac9e0ed9cdf31b06880465abcf19b42d56a10a4:action", "state_id": "09229234c428a7bf7009201f7a80452c818a5d43673e55a02139c40623b901a2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.19921875, 3.16015625, -1.078125, -0.91015625], "student_probs": [0.00453947763890028, 0.965056836605072, 0.013928063213825226, 0.016475500538945198], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c7a92d394cb00729deafaf81fb6e760367730af4186f408e2e153c48eef4ffb7:action", "state_id": "3b62959f4f386ff07a84f466c1fd75a432acef1320fc4eb6bb5dcc5a9e9c7918", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0390625, 3.177734375, -0.984375, -0.798828125], "student_probs": [0.005217293743044138, 0.9617703557014465, 0.014979256317019463, 0.01803317666053772], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4866783917d53b918b5e1e6c1dbbc8cfddc8c1a8d0c97dbe72cad575f545193b:action", "state_id": "cf0e8c61d52135f8bf44230a4d1a6c840703d51f111e3bfccca99bf77337f74a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.75390625, 3.232421875, -0.73095703125, -0.561767578125], "student_probs": [0.006515787448734045, 0.9538974761962891, 0.018122917041182518, 0.021463777869939804], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "36c079dd04840a4723e3d7009a725bac4a3ab52557ce6c33455965fa9c53f0fa:action", "state_id": "b3bca80562c2c20cb0f9fdd0f75d7eed945caf3f7559af916b2452027b91cb78", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.64453125, 3.283203125, -0.620513916015625, -0.4013671875], "student_probs": [0.006881494075059891, 0.9501028060913086, 0.019160542637109756, 0.023855144158005714], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "951056bd279e3010eb19e8ba177e74086cf95c9fa0d1f361c4af71ad7a4f900e:action", "state_id": "726255cc2946039bd08a2f5003f4e114d24504d17f34052d9b9003516ecb80c6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.71875, 3.255859375, -0.65625, -0.556884765625], "student_probs": [0.006588405929505825, 0.9532915353775024, 0.01906418427824974, 0.021055812016129494], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "00b10f3af63eae621f106fea09647f1953d5d56d0fa956fbdee246e8acd1b605:action", "state_id": "ad53ccf2bbd8131ead148d5f33eb21168bfa4819c88490a3d4be1d9cb6996f17", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.66796875, 3.279296875, -0.5009765625, -0.28125], "student_probs": [0.006711252499371767, 0.9448736310005188, 0.021558664739131927, 0.026856403797864914], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "69bd78381fc9356534c1fa76aa7c1c09dc930d9ca4bb2e3261d06ed51934ad97:action", "state_id": "a557a4ae5ac67dbbf40bc4fb94b06e201a3527adc2fe8adc31e48b3d33f58609", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.63671875, 3.296875, -0.4599609375, -0.19140625], "student_probs": [0.006785884499549866, 0.9424080848693848, 0.02201232500374317, 0.028793714940547943], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "03192be7599cb6eaf79a74cde9441dfb3874b0f6fa5f14c9be8a2bb8a7fe8226:action", "state_id": "2fdab9bfb3e5bd60f51ca60123f485e5c924b515bb39112f27ec97b32b3a3e7e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.45703125, 3.31640625, -0.34765625, 0.0625], "student_probs": [0.007878492586314678, 0.9322248697280884, 0.023891232907772064, 0.03600535914301872], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dc7cee430c888bd7524954bd960b7b8907a1bd7fc8f2a05cfd9af3ea04af609d:action", "state_id": "b1e099222462ed60c6e17074f26e21520804f48964ef8a53823c1019be261fce", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.80078125, 3.083984375, -0.6092529296875, -0.240234375], "student_probs": [0.007076490670442581, 0.9359328746795654, 0.023296577855944633, 0.03369417414069176], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6b2e2473bbceb6ac1028f2b82d30842b5f2dffd4a762428272ed4375257e43d9:action", "state_id": "dd85f8fb94dc2d5f6d2afc74542834294139be9084c7be359a3158a07c800a9f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8203125, 3.126953125, -0.633056640625, -0.212890625], "student_probs": [0.006664114072918892, 0.9382370114326477, 0.02184545435011387, 0.03325346112251282], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e24dfd9832cff4c653c6b877f3082635a338db3d3213f4e3f5a9a06dc57c3cb5:action", "state_id": "059b7756b6e57e8ee04b3c6c93a0647b0bcc378bb01677ad436a3143d3b494a4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.75390625, 3.15625, -0.44921875, 0.05859375], "student_probs": [0.006827201694250107, 0.9261823892593384, 0.025168731808662415, 0.0418216846883297], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "651f52ca1c376f2f1418d3d7b1049e0984ec5f3fff48eceb0d8f082424a0b92b:action", "state_id": "538643b3f248fefd4576d2cfa9c22e7d40a5c6503c755f6a413388a5548f3379", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.87109375, 3.12109375, -0.586090087890625, -0.46435546875], "student_probs": [0.006412086077034473, 0.9442322254180908, 0.023177646100521088, 0.026178091764450073], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a5ed76ee67632b2a839eb5d122d17264f268f0ddb67dc0c21f2fe534afd875d5:action", "state_id": "b36b2a39c4eededda91591a0eccf68fa039a87d4ea011cf918cd9b2b0251a8af", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7890625, 3.11328125, -0.599609375, -0.19140625], "student_probs": [0.006952573545277119, 0.9358504414558411, 0.022841179743409157, 0.034355707466602325], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b23d85d0f5af2b43255657cc7f03a7a2ad6531f6aa78b932247accc615f98309:action", "state_id": "a8ec359910abba03b076743ae8a83cbcf81f19922ecad0147a4e2a50cc0e6d7d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.83203125, 3.029296875, -0.639892578125, -0.294921875], "student_probs": [0.007238985504955053, 0.9352456331253052, 0.023846076801419258, 0.03366943448781967], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e5a9c975b43be62a47846e4a85f494ef7edd02afdb3c365edebe0019975ad208:action", "state_id": "171ccd68643ac4a2e77caf2b55c7b11f806e1d3cc00061fd5586a2809c4aff19", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8046875, 3.033203125, -0.3583984375, 0.06640625], "student_probs": [0.007249235641211271, 0.9148743152618408, 0.030789852142333984, 0.04708666354417801], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "74e5935f62b7c30723ef3910d1ca32e33be4f82d67c7e3831b5e0aa0646db3a0:action", "state_id": "30a6f4f862f6fcd48e7ae6979a5ae12fa84a8a89bf1c4ec46f919190bda12e35", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.65234375, 2.97265625, -0.263671875, 0.22265625], "student_probs": [0.008808002807199955, 0.8984407186508179, 0.035315874963998795, 0.05743539333343506], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "11160a34a66c26d5f7324a9ed1b7b08f5156ff8343a96c185dfd6559377f9d46:action", "state_id": "6048e61f9822bd7e78802d1be93a3f279b479a576606c507b1295c91bf0686ee", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.87109375, 0.51953125, 0.0, 0.44140625], "student_probs": [0.035068828612565994, 0.382962703704834, 0.22778594493865967, 0.3541826009750366], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1b7ae96ffa6020e381e447570cc1f206c81b6a6aca2f02aed7f1edc738298b68:action", "state_id": "c9c9b5563e67cbdb1295f2f15b2118f379b763104ad30aff75c771d65f289910", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.47265625, -0.140625, 0.19921875, 0.55078125], "student_probs": [0.056576598435640335, 0.214353546500206, 0.3011084496974945, 0.4279613792896271], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7d19ba26833ee4add46834f3e6e31e71e9a7d88202fa0479df58d11a0080d9f7:action", "state_id": "d4e56524100330411166edac9fcabd3079570d9550e18bb6a27a3d21353b58e7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.45703125, -0.18359375, 0.31640625, 0.59375], "student_probs": [0.05483010411262512, 0.19591422379016876, 0.3230079412460327, 0.4262477159500122], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cb18445bb33fae0f5ca2c30d6a09b145815abc7b31edb6a1db85e50f54955a6c:action", "state_id": "dd429998ed6f1ed5d3d4c24f8b0f036e55d9bcf3adddabb89528ba67394b914a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.50390625, -4.03125, -0.87890625, -0.71728515625], "student_probs": [0.1943982094526291, 0.015526758506894112, 0.3631836473941803, 0.4268914461135864], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1b7c4b9b4f75a87c81cf77d5404fba660ea824e09c5cf2e201de400a9b496197:action", "state_id": "7c6b9bace3f2d5e7f61cb3f6e0a1af5ab88243fb7c212ab8954347d9a41b9460", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -3.3984375, -1.07421875, -0.86328125], "student_probs": [0.30832499265670776, 0.029016748070716858, 0.29651325941085815, 0.36614498496055603], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cffb925f11259352f9d6f117b51329a01512c43ba03a7d220bcd0ca0b66154c7:action", "state_id": "a63de21bfec52c67a13bb1e847d70cf295cb28ab068ce5844f77a128f72e4ac4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.10546875, -0.6600341796875, 1.291015625, -0.77490234375], "student_probs": [0.16320431232452393, 0.09373179078102112, 0.6595034599304199, 0.08356036245822906], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dc9ffa732367a40551d96582440b370b762402a84a2ff8ade31a176bf7db770b:action", "state_id": "dacb348308500b5d6b1ec6e188f1a5bda98632cbcfc24786b4befd41d07125a4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0234375, 0.53515625, -0.607421875, 0.640625], "student_probs": [0.19786317646503448, 0.33006662130355835, 0.10528977960348129, 0.3667804002761841], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "abaa46858a2f145400bdd134f218a07f605c070d000cfc16aaa2fff829ed66d4:action", "state_id": "2eadc688c25b86f83721f443f91515dfe473165f60ca8d69319f5bcb9c342863", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.35546875, -2.20703125, -1.146484375, -0.318359375], "student_probs": [0.3776175081729889, 0.059282805770635605, 0.1712057739496231, 0.3918939232826233], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1618d8d414c45b05d28f7b3aa16ca66d58200d4f5644b6ea80d2ba93b6e6b5b6:action", "state_id": "0efb9d820690cb73c0346e3420f32f478c9ac7bdad34c1a6baea109e8b24d8cc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.39453125, -1.81640625, -0.68463134765625, -0.3984375], "student_probs": [0.33494651317596436, 0.08080960810184479, 0.25060319900512695, 0.3336406946182251], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "815755b71c9cd3f780da7e7478592aa4336fbaa0a808ae0630e9fcf8c1202ec4:action", "state_id": "065dc4ec8131fcbf0137baa6fa8dfe2407771a46ea21af3f067f258058118b3a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.62890625, 2.48046875, -1.2578125, -1.6171875], "student_probs": [0.0057717785239219666, 0.9556151032447815, 0.022738829255104065, 0.015874261036515236], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0b95fb42296c7f953906bd3393847b44a836f4c639abd071efb89da186b5fde4:action", "state_id": "e74d34f133d5621a3aeefc86a11e44063345c7eb2d9128960b73f538956e176e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.98046875, 0.375, -0.884765625, -0.53076171875], "student_probs": [0.05320218950510025, 0.5609143972396851, 0.15914292633533478, 0.22674059867858887], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7bc528177f590737653b2463275c8561c75e6e12fa96eeaa7a61c6314759139c:action", "state_id": "2dd8979407a1caaf050d227729d75ff94075103e1cb5785c2057ad2810aff6eb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.91796875, 1.05078125, -1.185546875, -0.6527099609375], "student_probs": [0.03832637518644333, 0.746121346950531, 0.07972315698862076, 0.13582904636859894], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2d8f87d1967ae955c2cbb8d285ce3293bca577fb30f744216fa529de4bfcb477:action", "state_id": "78b55f7732fe9a5e39bb4cfb5a6e9b5394aca55dd06f7038533a917284b6ff45", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.69140625, 1.48828125, -0.8896484375, -0.4013671875], "student_probs": [0.03236076235771179, 0.7779280543327332, 0.07214689999818802, 0.11756432056427002], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "328e634349bf5943ff75dcd4e30cd381457130f2f7b0e446832abb2174dce8df:action", "state_id": "2edae292e4a981c212f16ef65e630f0cabda8a47bf4059420a3193915698b15a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9453125, 2.87890625, -1.25, -1.1875], "student_probs": [0.007714434992522001, 0.9603636860847473, 0.015462315641343594, 0.016459548845887184], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81c85b13e66d666927303c93492e223bfd5518a40702fc2e99cd1c70867db90a:action", "state_id": "997ec9036c4548638f65a6d73f81747c594f91a20879d240ea886c4eab64df30", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.98046875, 2.892578125, -1.34375, -1.28125], "student_probs": [0.007373487111181021, 0.9638518691062927, 0.013937868177890778, 0.014836783520877361], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ea53823921e9da06317541b1d8055fad56a72dd43237bba32ba3abd2ac53eb27:action", "state_id": "bc9052b29db7aa0ba6a0b438cf5d6239c1bff02091ba8f2d08fc7d47d0e20a23", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9375, 2.8125, -1.3203125, -1.23828125], "student_probs": [0.008302188478410244, 0.9596024751663208, 0.015389826148748398, 0.016705498099327087], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2b7666b32a508059881e5154defabb98d8956c06e33f7af92894a9388a850a0b:action", "state_id": "d62585133f927d4459aeb1c02ae862f5e7f634dab5d760e18181be0698a612c8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9296875, 2.794921875, -1.2734375, -1.16796875], "student_probs": [0.008492138236761093, 0.9569490551948547, 0.01636902429163456, 0.018189772963523865], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d06797cff653aa284960a4ecf8871359fedd66de0990d35f4a1481b1b6918dd7:action", "state_id": "0e59f6fb5ed8156e0da97d9b87e15356a3c7963f932efbfdab76c4e7de690d33", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.09765625, 2.912109375, -1.33984375, -1.390625], "student_probs": [0.00645031500607729, 0.9667061567306519, 0.013762416318058968, 0.013080991804599762], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "abc2fbb86cec878de503640617557d8d0edf314c179f335491e270c0d387985d:action", "state_id": "e6e9de52f6046120b591cfb86761dcbd9c45305fff9c46e1633d307e9ff16cde", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.25390625, 3.087890625, -1.515625, -1.5234375], "student_probs": [0.004671672359108925, 0.9758550524711609, 0.009774710051715374, 0.009698642417788506], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "66b2622422ce69617086c4829a9d0403bc66c999ac154d30361d4c1109d6a4d0:action", "state_id": "d5f2b30dd1f0938b93d5c18b59ebd9e113aeefc0d49501a57a80e6c5af8e63ec", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.36328125, 3.09375, -1.66796875, -1.56640625], "student_probs": [0.0041732145473361015, 0.978203535079956, 0.008364520967006683, 0.009258680045604706], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cc95d1b9fd3e446264fd5a55a239e65efdf5db4db56024fce69e15c883980ec8:action", "state_id": "1493d9cb057cb2020accf4cbf477b21bc134270a6ba5c830ad1960599026d801", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3671875, 3.09375, -1.6953125, -1.46484375], "student_probs": [0.004153837915509939, 0.9774725437164307, 0.008132820017635822, 0.010240767151117325], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bba509b100f3a9f6c3b1451037d33f76db36b2a57fdad9497a7e20ad88b523dc:action", "state_id": "06b4c3265ebf452210b29d07b7e8c7b08eff175bd1780eca9add825cc5b5978c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.39453125, 3.109375, -1.67578125, -1.40234375], "student_probs": [0.003977746237069368, 0.9771319031715393, 0.00816180557012558, 0.010728491470217705], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f33cda7bd5e2fb8e6eef702eba64eb22059c3adc38d44cb2db0a9213a57b286f:action", "state_id": "55edd5a57afada479dbe62d7a918a12d17e544ba2f351a6b5f8851d6dc784792", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.40234375, 3.142578125, -1.67578125, -1.37890625], "student_probs": [0.0038199247792363167, 0.9776508808135986, 0.007899451069533825, 0.010629873722791672], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5ee78206686a0f50df0d4b0f9060605388ec478199d290bfeccf6ac70a82d7f0:action", "state_id": "5d4d9b46660bd9e4fee5cdd7adf58e2f102af4a11848da8879d95e5d91980a6a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.36328125, 3.177734375, -1.40234375, -1.27734375], "student_probs": [0.0038238991983234882, 0.9748525619506836, 0.009996230714023113, 0.011327213607728481], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "34c131fb3dab161ceaf7152ea935a097f87f711d42f769a6fd66c686526fa966:action", "state_id": "7af91864fcdf490fc99a7c4fdf13ea25cf559720084ac20ff40a2bfe510bcc2c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.35546875, 3.14453125, -1.4609375, -1.359375], "student_probs": [0.003986512776464224, 0.9754675030708313, 0.009751763194799423, 0.010794217698276043], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "150c16d23833402eef8cfb0ecce8d3c21372daf4a52f76782d95123d42ce9700:action", "state_id": "281f4d0754f108e5b4a3a008df1f93fcce9a5f83d7e2b39db4a950db9036adf0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.34375, 3.162109375, -1.484375, -1.390625], "student_probs": [0.003966910298913717, 0.976375162601471, 0.009368589147925377, 0.010289382189512253], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fa95356e4a3b515e2d1845dca010bbccfe184f8d7358d24e8bba7525e0513c5b:action", "state_id": "74584c9fefec0db7ab08e650e104ed541e8608b6361e77fad2525551ae2faaae", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.31640625, 3.1484375, -1.50390625, -1.3828125], "student_probs": [0.004131614696234465, 0.9760482907295227, 0.009310738183557987, 0.010509314946830273], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "303bf68b8ab79fa05aa07f5696a2805ef2479f98df25cdc1880c9a98864f1cbc:action", "state_id": "839859e6bb0f6d2172f07769f2c84cd59fc2fa2f211a0ca7930801805f614356", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2421875, 3.171875, -1.39453125, -1.24609375], "student_probs": [0.004336817655712366, 0.9737975001335144, 0.010122870095074177, 0.011742733418941498], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d3f74e50c75b424861506411d0cc7021d06a29a5b1ed45fa5786f2c5c62e995a:action", "state_id": "e31219f1e10f3a26c2fb783eb26c343355164cef4698d0c75667b59bba5f1d0e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1171875, 3.21484375, -1.23828125, -1.0625], "student_probs": [0.004691815935075283, 0.9705383777618408, 0.011299132369458675, 0.01347056869417429], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "64122c26a1fc38faaa674dfc25b771e2428b17d8103b2926cae15e6904d39231:action", "state_id": "fcf7d70f9f35b879052f1ee3fd662042c607ae2cdff8b10c538a6482f370eda0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0859375, 3.201171875, -1.1484375, -0.982421875], "student_probs": [0.004893822595477104, 0.9678557515144348, 0.01249681320041418, 0.014753632247447968], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8e231910a171230b8880499123e5717ed0903e52a713ae31feb3936e09e6a61c:action", "state_id": "76d41bf37bd21ff3d67cf6ff0f496e2eec15554ad8322988a7a8be0550d045b8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0, 3.220703125, -0.96484375, -0.7744140625], "student_probs": [0.0052005876787006855, 0.9624428749084473, 0.014642493799328804, 0.017714038491249084], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "74c68d6bc6d1a061bcc2236ea420509271aa27ab35bc34691ec0ccdb5c8dc4c8:action", "state_id": "3a6f65c9f9d7f2c6dcf93436a5d1196f0fd28b642216fefa3c0ceaddced7c3c5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.875, 3.26171875, -0.95703125, -0.7548828125], "student_probs": [0.0056584798730909824, 0.962827205657959, 0.014169956557452679, 0.01734444685280323], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cb345ffb853d1662a8220f6e2a3489f547be0d7f243fe5d6541b626efa19a6c1:action", "state_id": "6afc14a1265b74fa2ac300a59bed7371dbad4a46917b249e17608cd18a88f041", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.73046875, 3.27734375, -0.622314453125, -0.3583984375], "student_probs": [0.006347213871777058, 0.9493982791900635, 0.019224204123020172, 0.02503027766942978], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "291cdeec426190713580087de880000c99df9702a99a51893d7b1be4bdcb8d4a:action", "state_id": "638bf5d34772585d3250980d0a9b2e602ff3abc96c6be14fa6de51f3625cc717", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7890625, 3.146484375, -0.5879058837890625, -0.166015625], "student_probs": [0.00673211645334959, 0.9367687702178955, 0.02237728051841259, 0.0341218002140522], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "82a0308b7882028381394a72dbbb8635765dc50b7f5f650a2a6a92108c8d139a:action", "state_id": "ea20b249826488ab658c31f5dfabd86a14ce0b61ba1311b017999e114de37c0c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6953125, 3.150390625, -0.4873046875, -0.125], "student_probs": [0.007334193214774132, 0.9328557252883911, 0.024546155706048012, 0.03526390716433525], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "86b3205ddbb338153d538e2ca76a4f9d1ec1f3357ad46a1ec676e37a9be58969:action", "state_id": "aa42652e6c74617b052f68f18b6d5a864b0b7ae2176faeac00d0d7b487e35b02", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5, 3.15234375, -0.30078125, -0.046875], "student_probs": [0.008816447108983994, 0.9242315888404846, 0.02924877591431141, 0.03770316019654274], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3f3e36c4975433d875b53437efa38add43f527bbf1bff063682080f0d782f016:action", "state_id": "50058836d36cd710f1e8d34621d97d31f8791442660e7fe3f0ecc1be47d56ae5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.390625, 3.171875, -0.201171875, 0.02734375], "student_probs": [0.009593545459210873, 0.9192797541618347, 0.03151752054691315, 0.039609115570783615], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b9f12a601975b9c936aa43fa8ed453e4252a2a02f678b7254cb8837562826e7f:action", "state_id": "0491eb940894e591585b42407ae4292ee4deabf1d6ab5c41ad26bfbace092a8d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.34375, 3.138671875, -0.341796875, 0.14453125], "student_probs": [0.010351786389946938, 0.9156011939048767, 0.028194082900881767, 0.04585298150777817], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "daf01ec62a42f72acbecf9b018a4ab017940edf4f7ccf6039ee23becc8705db0:action", "state_id": "53b5ca522d097c6225b892147c0936b86afab43d2124e97032d46183f6344def", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.51171875, -0.05078125, 0.1640625, 0.4375], "student_probs": [0.05657336115837097, 0.243831068277359, 0.3022696375846863, 0.39732593297958374], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0bae7add0ce0fac22667663aa12b023348c0b57d29da5b1492c85f9bfec772cb:action", "state_id": "4482a61bb06ff71f397916408e8366a6b844d0905261a2f71eb02ae8daf88de6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.099609375, -0.345703125, 0.34765625, 0.5234375], "student_probs": [0.080351822078228, 0.1707705706357956, 0.34161362051963806, 0.40726399421691895], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a7bba5ccb0a676f63229c32caa3e979a4513f0acada2b7bc5d3d9ca00d07d288:action", "state_id": "56b6ea37257e0305d0e0544a9abe1054351443212dbb4aae9b4a6242492cb975", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3359375, -4.0, -0.7646484375, -0.6396484375], "student_probs": [0.08729135245084763, 0.016530198976397514, 0.4201200604438782, 0.47605839371681213], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9d7314aea870aebf1acf550172d1ab95ede2e55062ec6bad1744ff56c11f0884:action", "state_id": "f9421b03b4ca8f44c757ec3c0ebc79ba7dce758358143e87782f2051605c8026", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8134765625, -3.421875, -0.853515625, -0.6021728515625], "student_probs": [0.3058392405509949, 0.022525794804096222, 0.2938356101512909, 0.3777993619441986], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "783339fa100f39501d386a109349088186dee12e9070da43a4c21b52597962a5:action", "state_id": "7d7a8eb530f9df6b87bd6532c3d94886391898ee48e939a8ab10820f07705571", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.38671875, 0.1796875, 1.16015625, -0.3876953125], "student_probs": [0.11823522299528122, 0.2083214372396469, 0.555323600769043, 0.11811980605125427], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dbd1daee1150599de4827eae3cffded2aa1b970a83137d7eb444d5a0c35dbbeb:action", "state_id": "61649df0615c2ffa7133feba986c1b8e8a3b24f2b8400d073176afa18faffb0f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.806640625, -0.31640625, -0.16015625, 0.79296875], "student_probs": [0.37146997451782227, 0.12083441764116287, 0.1412697434425354, 0.36642584204673767], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0d9eb7aa60bd219017cc9f05e53419e502d158e9b68bff29730bd014c2cb228:action", "state_id": "14164ed42ea11e1b2cbdf73ca53506b9fbc23cc70030a627eeaaeccd6c94464f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.525390625, -0.185546875, -0.71875, 0.45703125], "student_probs": [0.16949638724327087, 0.23809634149074554, 0.13969650864601135, 0.4527108073234558], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fda80fd58b9a8372cbdc943757531c0e8a211cfc254695c3d36fb5edcc5706c2:action", "state_id": "8775db838b971e4c2a0e5091c04732ceba1d5f2c38a1fadbd2d88acc4d84c055", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.15625, -3.01953125, -1.21875, -0.388671875], "student_probs": [0.23534297943115234, 0.03651644289493561, 0.22108426690101624, 0.5070562958717346], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2cc3b58fe12ab0447a1e6f0120a2e9c93a68b2bb147bf26632914539916f4b95:action", "state_id": "31a7b295a1fab14936bd0e520e23a3bee87617da455099d673a37958edf0f8bb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7109375, -2.78125, -1.16796875, -0.751953125], "student_probs": [0.17627184092998505, 0.060443855822086334, 0.3033830523490906, 0.45990124344825745], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3bcb6fbd8fb5c62b8d2f3fc19e8688562defe9ca2b249ee9a83dd6122832c205:action", "state_id": "54ce4033d4ef2e3acb4dae7fb9ed0c5c5483990133a07d79ff410b025b782ef2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.34375, -0.6600341796875, 1.091796875, -0.5126953125], "student_probs": [0.14759385585784912, 0.10757411271333694, 0.6201808452606201, 0.12465114146471024], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d04ebe79c88a35b75a32a5efed25ef8266159508209fb76b77c5485a4d60caba:action", "state_id": "cf660c1eef5ac54d05c182597c7a1a020ae8d563acfc1f0b73358db9a071f876", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.349609375, 0.03125, 1.93359375, 1.1328125], "student_probs": [0.2586762309074402, 0.06921502202749252, 0.4638501703739166, 0.20825855433940887], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "156f9589826168dfcb8166609a3a9a5ab102477e74c74eb3a960950686b6c0da:action", "state_id": "f271f79942e03fddd58484c4e86d476b5e6789ac1f74985bd4ad2a3a24df4662", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.478515625, 0.611328125, 2.037109375, 1.43359375], "student_probs": [0.24245858192443848, 0.10186448693275452, 0.4238690733909607, 0.2318079024553299], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "89970f2659b454a8709ac6ac53becfec177b8093706d017d4f93ce15d74901dd:action", "state_id": "92578136d3f08f7fd7d299e2c84f846ee3be08137bf763054279ad4a628bb07e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.822265625, 1.0078125, 0.08203125, 1.302734375], "student_probs": [0.2326820194721222, 0.2801204323768616, 0.11098980903625488, 0.37620773911476135], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6751878e8f2e52630ce622499e7ce04f51b5341578e714180ef7db4f5e92e493:action", "state_id": "590b8691780c2638ac5c18c4b6e5f993326695ecb1a5cebfea8945c8d63a1912", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.388671875, -1.63671875, -1.078125, -0.60986328125], "student_probs": [0.38602903485298157, 0.1108153909444809, 0.19372883439064026, 0.3094266951084137], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d97a68bfd00181c6cfab3e58a895ef75d82b194ace14fd1f287cc3951c8dd2fe:action", "state_id": "8673264df8876e6e7ce97297ce1cd49f887c672024b6a967172dabda8cd05b19", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.63818359375, -1.5703125, -0.7003173828125, -0.83056640625], "student_probs": [0.3166097104549408, 0.1246538758277893, 0.2975362539291382, 0.2612001597881317], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a73447fdcb6e3208b3a07fe753eea95cc0b2166a212c755edb87e6808e755a3e:action", "state_id": "e4bb6177c019d3fe2a6aa60fb03774564e35db360d9e467354e6b92dfd5dec8b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.056640625, 2.169921875, -2.01171875, -0.984375], "student_probs": [0.036162927746772766, 0.9110492467880249, 0.013914846815168858, 0.03887300565838814], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a3594287e0661d3bd415b9ab7e5c21056d35ffc7329cbd710d3662b76483ba57:action", "state_id": "271dd5e3d21bc044d67eea0c0cc2a4ce918d64bb86e235e0aae3ba3d5870781b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.10546875, 1.01171875, -1.6171875, 0.51953125], "student_probs": [0.1935521513223648, 0.47904616594314575, 0.03456669673323631, 0.2928350269794464], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7aa95afdae22878a7050376ddf96e3f382ceffcbcbabeb9bb23ba04d0f34a6fe:action", "state_id": "8520c013e7a453f2980a23cd91c59a22689e2564ee8e3fa6bccfa742754b7425", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.641357421875, 1.48046875, -1.671875, 0.24609375], "student_probs": [0.08242583274841309, 0.6879561543464661, 0.029411371797323227, 0.20020665228366852], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8f4e53986e2899b73c85d7d5607578ee307b4ba64ab3605415588865ea449c7c:action", "state_id": "5ee4bcd7900ca02f1cb99411fb057d4701a908197cc7ba377646c8571f5fbc7a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9150390625, 1.96484375, -1.54296875, -0.0234375], "student_probs": [0.04590332508087158, 0.8176385164260864, 0.02449839934706688, 0.11195971071720123], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3580f8c9418205e42a2f8a4dd77988edffdb1bbe034386a6dc12fcc8f6e1a447:action", "state_id": "0e21cbb0ca1ddf82d5b9fc76cced01a82d18ada6eb7ab6dadfe315e86562024f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.39453125, 3.017578125, -1.99609375, -0.7958984375], "student_probs": [0.011653549037873745, 0.9607557058334351, 0.0063856178894639015, 0.02120514027774334], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3d29c8e8d7fc2bb9961ab54d3cbaba1dacfea9bde3d3e5a035a8ee385d601c87:action", "state_id": "1e87dd3cff28fe158eb8bf5b87caeba81cbb3ad75c9f245adc4d5f402f58bd94", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.08984375, 2.96875, -2.359375, -1.4921875], "student_probs": [0.006213085260242224, 0.97774738073349, 0.004745165351778269, 0.011294477619230747], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "41ca044a9f9a29447957bca56e09429214a5c119c5788b805854a3a18d163644:action", "state_id": "ea5c23f51064d755aa73f34fde9aa7c81e265056c3e66a0e563f51706ade46fa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.18359375, 2.990234375, -2.2890625, -1.69140625], "student_probs": [0.005551690235733986, 0.9803705215454102, 0.004995980765670538, 0.00908195972442627], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "45b87cf13d690d3994935bb786590148b134d7c351fa1cae338e3b837e9d2330:action", "state_id": "e46ac5a9487b7ab52117bfa84233656ffbb02bb12b53f2737208961af9188713", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0, 2.916015625, -1.82421875, -1.375], "student_probs": [0.0071165128611028194, 0.9711039066314697, 0.008484144695103168, 0.013295397162437439], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "35ccab39a754cf0d903e5e432bfb4871bce3d16c2e3437abb0ae1c308ff3190d:action", "state_id": "10504eaa5d928bcb393add103250e9f53bf6eef778349971bd8cd81f24228605", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.12109375, 3.08203125, -1.9140625, -1.6328125], "student_probs": [0.005385054275393486, 0.9792161583900452, 0.0066237300634384155, 0.00877501629292965], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "355a9ce773c614be5bb6e01e8e254899fc1fa9b122ddd804f78183043fe8543c:action", "state_id": "b097ad4c2dd03165d917e266b76a2a47e234121192c789ced6968137892226aa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.40625, 3.02734375, -2.10546875, -1.80078125], "student_probs": [0.004289017058908939, 0.9820589423179626, 0.005794092081487179, 0.007857954129576683], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "15f1773e924e57eba8b0b7199b5f716cb951d71fabffc3d75112ad2444e17bac:action", "state_id": "454120b2b041c3ca2943f7154d80246ec65bbccf916dd9508b41fa2c453b998a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.40625, 2.9765625, -2.234375, -1.82421875], "student_probs": [0.004512417130172253, 0.9820531606674194, 0.0053586275316774845, 0.00807573739439249], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "569ab5b2f1a23910a95f903ac80bbf82ab061af8b1f74aa2aad9b0607a1eaf0f:action", "state_id": "bccaa920c4cc60d63e3763897c8c9bae12cefd9ae88fb62094e2d20ca1e2e9c6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.35546875, 3.078125, -2.07421875, -1.75], "student_probs": [0.004289497621357441, 0.982168972492218, 0.005682661198079586, 0.00785883516073227], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4089d13222d13998a72d7ff6c2e44a548b92076c3d72c1e6e7b2d295a51beb51:action", "state_id": "8814a405e7d110c4c5e65e0c532ce0742ed9564ef2f6ba92821d68a6522d55ad", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.46875, 3.16796875, -2.09765625, -1.77734375], "student_probs": [0.0035089377779513597, 0.9843997955322266, 0.005085569806396961, 0.007005668245255947], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "da3fd1cf99bc7a2f7a1609a67592329deebb20ab4ef94da3913e5a2f49723d56:action", "state_id": "f52b550876430f0ebb172e532d3177d2c523f8eb370908b7344b3ef18e46e5e9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.46484375, 3.18359375, -1.98046875, -1.703125], "student_probs": [0.0034648505970835686, 0.9834895133972168, 0.005624007433652878, 0.007421552203595638], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "39dc410c3d4b6d9d365d37cdf996b392aecdebe1ded967028b591ece30798485:action", "state_id": "2c70ac8f9745d3575f785ad5fdeaba0a51eac73f136d99ca95dc2c5ef96e5ba9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.48046875, 3.23046875, -2.16796875, -1.86328125], "student_probs": [0.00326397642493248, 0.9862241744995117, 0.004461326636373997, 0.006050456315279007], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6c4c366acf11275d340811ebeef9e3ebd5add8cd90b1fa7d94f97cd995d79029:action", "state_id": "1a06f3c94e5017f11e36f77df61d7675451b1cb7cb8883cdaf1f3fba480afaab", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.50390625, 3.1640625, -2.26171875, -1.9296875], "student_probs": [0.0034072035923600197, 0.9862014651298523, 0.004340889863669872, 0.006050317082554102], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0ec7071476faff74b56f3389e04c2674e1bf588945bc9d4915b3f01db2b8226f:action", "state_id": "47531dbc50bea3c26856e5d9b9ac5f5c42ca63e2fa999ae8442bcbb028be5c0b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.5625, 3.1953125, -2.16796875, -1.94921875], "student_probs": [0.003115409752354026, 0.9865097403526306, 0.004622297827154398, 0.005752542521804571], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "89595e88e710b0b59acaa86fb140de9492ef2f5273d092fbe38c5f5cc2ce83cf:action", "state_id": "82ef880105c12df1be7075069e80fecd0209b5b43fded6739ada56669173a605", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9453125, 3.130859375, -1.9296875, -1.68359375], "student_probs": [0.006117155309766531, 0.9797222018241882, 0.006213486660271883, 0.007947170175611973], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "692d9bf386a4ca0397fdd714cc459e057ea113250cd42f8df4ffd93d5820a6fd:action", "state_id": "18f0610656f4a6a8802cdcb4d612e46d5798dd528bc769b9b4a71be39edb1ad2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1796875, 3.14453125, -2.03515625, -1.83203125], "student_probs": [0.0047888318076729774, 0.9828978776931763, 0.0055334847420454025, 0.006779766641557217], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "48526f76cd059550ff8a1a92d1a6f27b04f60640683a4a7aeb93b38f5c136541:action", "state_id": "5b85b27b5f0c402c6190af6e8b384f424df235869e84306f7b64b709ec639ffc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1796875, 3.193359375, -1.78125, -1.64453125], "student_probs": [0.004551336169242859, 0.9808971881866455, 0.006779194343835115, 0.007772384211421013], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "071abba709b0ae5a5083ea015754feb8cb00b121b26183430e432dd977f7a3f2:action", "state_id": "fab90b6a3abc310ef3f441cb1a5b8aa5c771e0ba7153c3da268e440a92bdd1b4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.08203125, 3.236328125, -1.50390625, -1.53125], "student_probs": [0.004794642329216003, 0.9783411622047424, 0.008547374047338963, 0.008316823281347752], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "37466b9efe715064da68d9f4cb82e37917310e0a6ed537fe22a55242d9b52bf2:action", "state_id": "70e9d03204dd4d5bc644c6e225621251809575aa284f5e0b61f8c703f238a7c7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.54296875, 3.22265625, -0.81640625, -0.673583984375], "student_probs": [0.00813948642462492, 0.9556121230125427, 0.016832130029797554, 0.01941628009080887], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8ec9fd7f92e91dae78990c6f4ac36fa1d775f41401aaa8471eda8ff978644b5f:action", "state_id": "762dca64ccaeaa79137436b8a9fba0515bbc18dc4fe89d4b265fb797891389d7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.62109375, 2.9296875, -1.75390625, -1.1640625], "student_probs": [0.010187306441366673, 0.9648028016090393, 0.008920303545892239, 0.016089608892798424], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9b1854219ff198996a193e0b23bd3aae303c0ebf7e40371824611410f98735ee:action", "state_id": "19923ff6a06c5c643ba89a6aad5564b8a117c99be67bad83526952812527dc12", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.83984375, 3.001953125, -1.59375, -1.23046875], "student_probs": [0.0076443771831691265, 0.9685181379318237, 0.009777307510375977, 0.01406016107648611], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d90b96eb86a9074eca7f1f561894e82b9487dcdc2501565b623527dc311d1829:action", "state_id": "a7065b224d3ac968550f1028bced24548b884bab215f0dfb3e92e4eba38bb8fc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.80078125, 2.396484375, -1.8203125, -0.03125], "student_probs": [0.035733357071876526, 0.8742358088493347, 0.012891307473182678, 0.07713952660560608], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d81542bb4e0cae7db816ae645a9830bd2788276a6340cedaac1eb965553ba004:action", "state_id": "7f8b4af4c0a21c2732516e8b6c010f780b67ec9ecc471047cd1405db57e998a6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6982421875, 2.646484375, -1.9765625, 0.03125], "student_probs": [0.031540412455797195, 0.8942597508430481, 0.008784153498709202, 0.06541567295789719], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3b7a388a9535bfb41894bae188542398da1033409902b69211aa8e0ffc0e60ab:action", "state_id": "3ef5cee109a645d2b5b2be747d8d46aa86c6b0539d7a5b8e9e89be7e1ca88f9a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7587890625, -3.921875, -1.7734375, -0.818359375], "student_probs": [0.42607688903808594, 0.018020929768681526, 0.15446560084819794, 0.40143656730651855], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a1d6f1398336482ba16f4fe7d710f2c70d3cfc90518cfae7cbbc9560736391b1:action", "state_id": "c349ba9075836f724cdf39fe754c2f55b16060e126a42b30852c2dafbbcf3d75", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4111328125, -3.0390625, -0.7666015625, -0.628173828125], "student_probs": [0.387902170419693, 0.028017336502671242, 0.27185922861099243, 0.3122212886810303], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f42073fbb2ecd31f70b6108e37b7a0b369701ef1394df526dd9e8a16bd23e73d:action", "state_id": "50a3d9e989378c3594a5bca1ad967656e9b126e5758021d262423eea4a77d0e8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4375, 0.1953125, 1.076171875, -0.4619140625], "student_probs": [0.11901696771383286, 0.22409690916538239, 0.5407396554946899, 0.11614646017551422], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c6c7d9109f53ef2c42708166dacefffca0a9654971c4f54d5e7d0da1deade600:action", "state_id": "922d1f488f9f8be4b4b26b43029c7c1ca938585c559b868435a805692cb5a1d5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.419921875, -0.0234375, 2.0390625, 1.1953125], "student_probs": [0.2569179832935333, 0.06066685914993286, 0.4771817922592163, 0.20523333549499512], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b226e7e2211eb8fb7084a89896b05d25c4ae2768afe2ad06be6594f70ecb21f9:action", "state_id": "3327a4f420126213c1964f03c12a3aa7f16ab29b394c554fef6777d86b05f20f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.958984375, 0.33984375, 1.857421875, 1.158203125], "student_probs": [0.1917685568332672, 0.10324951261281967, 0.470938116312027, 0.23404373228549957], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "86c3b68f8c489341cff1dd6daf94aa72a2a4e4456cc3ca604fc5be258bf2b7d3:action", "state_id": "5b263edb703475bce54da7a3c8c41dcb7ad89ef22276c4ebb2f25352090640b5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.38671875, 0.974609375, -0.453125, 1.14453125], "student_probs": [0.1863735318183899, 0.33550724387168884, 0.08047199249267578, 0.39764729142189026], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a610179d26450b1e387e777d6a19868aea582fdfad164b886c8ac3856058073e:action", "state_id": "c36177c525acd109268ddb6cca53182822910f14dd3d1fadd4ce8a3b6eae691e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4169921875, 1.984375, -1.63671875, -0.134765625], "student_probs": [0.07320832461118698, 0.8080923557281494, 0.021619217470288277, 0.09708002954721451], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a9a6e625499384d18fed3d92ee59405ca56facce7c403429bcdfa5a25beee054:action", "state_id": "2949ba6b8a441ef50a8320a5b53785c1ab18aa2653f5f9ea3c0484383e19a4bf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3515625, 2.44921875, -0.779541015625, 0.546875], "student_probs": [0.09358546882867813, 0.7624457478523254, 0.030197875574231148, 0.1137709990143776], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "29c65d3c373b1989c4f6362e1150318674d020c87e1d57b311416d4101e13187:action", "state_id": "d32f21bfbb9fb9c8f3c259cba334cc62d03325cf6807f16cfbc0d000d6d87e60", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.0546875, 2.380859375, -0.826171875, 0.04296875], "student_probs": [0.07149510830640793, 0.8166216015815735, 0.03305406868457794, 0.0788293406367302], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a1fe861a466f06848959b83e0a22afc9626409111fdf67022556ddb031d87268:action", "state_id": "0f90db3f6b6a46d2885d2eb4334bffef172d0202cfe1fa5fcf9a2d096ccb20fc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.5126953125, 3.044921875, -0.494140625, -0.486328125], "student_probs": [0.02622953988611698, 0.9201194047927856, 0.026720764115452766, 0.02693033777177334], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "37693aa4f98ba85f2e03e182d83ffa2d17692c1d35f805ce46df9214b46651b1:action", "state_id": "6532b7508002551d73e9b7ca875dc3aba828b44a6dde93b4de8c9d53e065a995", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6162109375, 3.1669921875, -0.1015625, -0.4912109375], "student_probs": [0.020936790853738785, 0.9203104972839355, 0.035028304904699326, 0.023724494501948357], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "682d35d27356596bed2f4eb2d216345faeae025f85afb33813e55161d8ef48ac:action", "state_id": "103333bf53c6555651424c2d1e89c148075793f24cf3e21e517e2cc7623ccd73", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.0703125, 2.716796875, -1.32421875, 0.25390625], "student_probs": [0.0529034361243248, 0.858835756778717, 0.015097997151315212, 0.07316279411315918], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "479bd00eb8ec72f329fb08c0d7ca5fe1fc8be58b20671a107541db4f262834fc:action", "state_id": "63684295c75ec385468bd82da0b73ec8644b7605b2d7033528f0867817de9b9d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.28515625, 2.814453125, -1.6171875, -0.0546875], "student_probs": [0.040465496480464935, 0.8979002833366394, 0.01068048644810915, 0.05095375329256058], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "19c18ae2cf42c0ee0b0174e5d5e203c29a92be68458bc154ccf95247534d0b5a:action", "state_id": "e923c31fac6f242fd24757de5bed1a1fd0f02a59f0a1337493f686c6b40f2a40", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.63970947265625, 2.923828125, -1.58203125, -0.13671875], "student_probs": [0.0260884128510952, 0.9206029772758484, 0.010167227126657963, 0.04314135015010834], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2904097b032385d2e2d7558fbbf49d88b3793968147975df1d3c3dc8bb513554:action", "state_id": "a7f819b187585f373291c186abc752d5eaf731c8b2554f69fb76745a24a3c811", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.576171875, 3.080078125, -1.51953125, -0.177734375], "student_probs": [0.024041524156928062, 0.9307889342308044, 0.009359793737530708, 0.03580974414944649], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e9239cbf7acf2c301ef345fd9c1b7fc3d8105d587286cee04c7067a1ded2cbe1:action", "state_id": "84f471f4cc1d7f93fc45e660d5c7c078f961b81ade7e08a53e35d7220349a040", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.306640625, 3.185546875, -1.36328125, 0.06640625], "student_probs": [0.028044573962688446, 0.9214814901351929, 0.00974890124052763, 0.040724996477365494], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "807bb10e25e3eda91276738fde8cc78e3daa080bb8b19b8e8c09d81f34d3b54e:action", "state_id": "f06a85a07c25835142bed514a76a86f1fe95823d3c4451da4ec1ebff50f66590", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.83544921875, 2.962890625, -1.01953125, -0.5634765625], "student_probs": [0.020933005958795547, 0.9341779947280884, 0.017413489520549774, 0.027475640177726746], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0cc63b0e44112b5b91b778d50834c1d618124b63148754a3238c6e4133081706:action", "state_id": "aac9cdbc835c1d86acae8c5a20ec530a8f1dd28889f274fd2550c7932c8fb507", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8603515625, 3.1982421875, -0.716796875, -0.7122802734375], "student_probs": [0.016338052228093147, 0.9458562731742859, 0.01886015571653843, 0.01894553191959858], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9297fb0794a1faef3cad57ba72253ecf215ab6e901dfa70289ef3f70f7ce9ef4:action", "state_id": "5471c782d4e75c82d20902f8ac6fa07e5cf617a0892d6d71eb5b86608114f254", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.125, 1.771484375, 0.234375, 0.0234375], "student_probs": [0.12183663994073868, 0.632174551486969, 0.13591860234737396, 0.11007023602724075], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "49502faad372b3c071432a1a154cacad01c0c8a9f89952d125746459203ea692:action", "state_id": "a472131837096204d003e9c74478a1fa9dcbc36093b6bcd7763ef7bff25240f0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.19921875, 2.353515625, 0.44140625, 0.0078125], "student_probs": [0.08531218022108078, 0.7355467677116394, 0.108690544962883, 0.0704505518078804], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b48ad523fb639fe6e586f0b05ff339fce6b8b6e3a12253004dd9965ee3beb4ef:action", "state_id": "78e854919c936ecc9448aebe08b2588834a29f4bfb5b4573ad39cdf1ff5a62b4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.5849609375, 2.33154296875, 1.6875, 1.259765625], "student_probs": [0.20242327451705933, 0.4270678758621216, 0.22428105771541595, 0.14622779190540314], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3fecc190549b999e51137166e385df921f884202bf15fa9a6d0e6048972ad613:action", "state_id": "c086c9d3bd8d039ee9fae8742f90c5b76ec7e20fb5a7ad8e0b48f693efe04a4e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.091796875, -2.765625, 0.06640625, -0.05078125], "student_probs": [0.304668128490448, 0.02101833000779152, 0.3568894863128662, 0.31742408871650696], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "137f9bcfaa36f942e700ad0e94cfbd79c99c27eefd126bc8a7ec71e0fb514d29:action", "state_id": "d528c9762aea71110ddfd7a0cf0549045d057b8bf61e90bda431c50a1939f9c7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.080078125, -2.27734375, 0.15625, -0.115234375], "student_probs": [0.2991190254688263, 0.03323408588767052, 0.37886112928390503, 0.288785845041275], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "146ba54ae663bc96a77a7681c8d082c37e48627f1d238b44062e8cde3061f279:action", "state_id": "f6937e3d499b22c2cd43a260253d87b10af15c1364dc3db10e3a7919d2f91fbb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.00390625, -0.6833648681640625, 1.330078125, -0.572265625], "student_probs": [0.17037273943424225, 0.08636046200990677, 0.646758496761322, 0.09650832414627075], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a0a9dd4bad5429433597a7319c20b9c3043e6c8a823458f4e718a14e30cef1f7:action", "state_id": "6b22f0ceb1389cab3a401099368a0355a3592c3318e0da0d2c7f9750146f6ef9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.01953125, 0.59765625, -0.90185546875, 0.66796875], "student_probs": [0.19633986055850983, 0.3500136137008667, 0.07813673466444016, 0.3755097985267639], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "32f0d0145d3ab060e7d99b4c50526f01b0aa88a61c09ae538a288ff72a0a2b6d:action", "state_id": "c4c8d1f25f25c9ee07663bbd29a3df3ee6d3cdaad184b63ec1c58a59e72d5b64", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.30078125, -2.44140625, -0.9609375, -0.287109375], "student_probs": [0.3776267170906067, 0.044401854276657104, 0.19514639675617218, 0.3828250467777252], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7e0813829d563b9aef0f0209e46c07edb2f1b79bf11150cea5b3fe7b7509f0e1:action", "state_id": "87d49ccd8cdfe99ef62322345da25b0bd6857c934241b0c5f8360c91284361ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4599609375, -1.97265625, -0.78076171875, -0.55517578125], "student_probs": [0.35025525093078613, 0.0771666169166565, 0.254133939743042, 0.31844422221183777], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "109a841965b3a1ddf24945999b4e9fb8e4692b31dbb06c64c21918641ac8ab7c:action", "state_id": "7eeadfcfd98ec3754f95dd5eae824b0300583523fe734529e3a465ff929f7359", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.51171875, 2.484375, -1.064453125, -1.39453125], "student_probs": [0.006404416169971228, 0.9467939734458923, 0.027228204533457756, 0.019573472440242767], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fe0b390c0d4dfc4e1846b010bad547d4cc91cb9aa469b651a07c95f66a5b1442:action", "state_id": "c4f3710a7fdd93fb25369abe0a985a114e8b24089947ca329d2a71b287a8fb37", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.875, 0.390625, -0.7294921875, -0.330078125], "student_probs": [0.05414539948105812, 0.5218071341514587, 0.17023517191410065, 0.2538122832775116], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "97907ea33fc39a2173c1411ac0a534bccba641da2fadbd1c37062b06c6c1ce18:action", "state_id": "f019307d4c79041ce5cf7e36d32984aa8837775ccdda6bddc1b8c9f240f4cfeb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8203125, 1.1328125, -0.84765625, -0.337890625], "student_probs": [0.03674537315964699, 0.7042527198791504, 0.09719005972146988, 0.16181184351444244], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "84085c2631345298ec3cf9123da245c816975ec4de1d6b70dddb59eca95596b7:action", "state_id": "7ed250bbd7fd5fb89d9622fc5b2b6929e1d737c3e08baaedeede2f5bb673c56f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.72265625, 1.57421875, -0.7255859375, -0.291015625], "student_probs": [0.028633657842874527, 0.7739117741584778, 0.07760665565729141, 0.11984790861606598], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fe4bba68b1f78b1d9edd379d04f73c3a53dd9d033bc62f06c24d1b2dcfcb1445:action", "state_id": "604d662f107af8f5329d0e7e8f1fddf75dbef2b300551c91f2db541d4e5085b2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0234375, 3.044921875, -1.3125, -1.11328125], "student_probs": [0.006081467028707266, 0.9664266109466553, 0.012381252832710743, 0.015110687352716923], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ebad2f34e59a2df67ea7837ff488724ee22ff362f2dbbbfa141e809ef7f645c9:action", "state_id": "13d7c9764bbf7796c27e10a4db0ae017ca9f6177a48f936153e724ead99337e0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2109375, 3.07421875, -1.51953125, -1.3203125], "student_probs": [0.004930523224174976, 0.9732114672660828, 0.00984389428049326, 0.01201396994292736], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9c6972026c0246ee93625f5644006b64e1839c12d8eb100f7227fbb1ef9f46a4:action", "state_id": "c04fff4158654ec691f07f187b05ec62affc51e70f841406120e03d69d1c2f2c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1953125, 3.05859375, -1.54296875, -1.328125], "student_probs": [0.005086149554699659, 0.9730421304702759, 0.009765589609742165, 0.012106089852750301], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1099b9022ada7825da18fe80958826c9198d0df80605fc7d419b6a6d69de89af:action", "state_id": "fd1a535252c8dbc4c5af700b30e51d68ec3d6a636e01c8c7c759fd1dd8bb4942", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.359375, 3.0625, -1.67578125, -1.44921875], "student_probs": [0.004314646124839783, 0.9764175415039062, 0.008547245524823666, 0.010720647871494293], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fe0624af38f4cf00a0a6c2745c1e058883ba7bab84053656b1969cc5f5c2bb14:action", "state_id": "9a0c7f49ec4957215c6bfb1d7b984f910011763b6a4303c3296020585d893956", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.32421875, 3.05078125, -1.46875, -1.3046875], "student_probs": [0.004503200761973858, 0.9724206328392029, 0.010593676008284092, 0.012482400983572006], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cab6031b733dca9e41060e99e49c1a753ad434c91389c6741c95d9946cdf7c44:action", "state_id": "0401f20998d36e063255a263c77916964aabf82e13c97354c6d50eb8209f2be6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.39453125, 3.126953125, -1.57421875, -1.42578125], "student_probs": [0.003907597158104181, 0.9769222736358643, 0.008874972350895405, 0.010295148007571697], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "760e4fccfb9d46bf13768efe01d0831629f9897dcd66daa4af3c09849aa14f70:action", "state_id": "0b0f4e9eb815fbacb71b61459dbad5d315be4b1e441bf22a205fee25b6d383ec", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.46484375, 3.0546875, -1.71484375, -1.50390625], "student_probs": [0.003917739726603031, 0.9775468111038208, 0.008293855004012585, 0.010241544805467129], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "032a0421c184ddf20d01832b43c3c919bc3fd4508a5bd47956c940f36a3a46cc:action", "state_id": "159bbf168d784104d9704753a4ac0fb9dcecd161663962ab7c6b20ac7a8fac59", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.44140625, 3.05859375, -1.7109375, -1.50390625], "student_probs": [0.003994861617684364, 0.977510392665863, 0.008293545804917812, 0.01020123716443777], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b449ba3b82e5469c6bce103c75e0db164edd621f7bc1386ca61ce9bee3fb4a9f:action", "state_id": "5e812987fa395b00251e7f451040d249056b8fdf9d014ffea7719ab324e3fbf7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3984375, 3.09375, -1.5390625, -1.37890625], "student_probs": [0.004017334431409836, 0.9753594398498535, 0.009487674571573734, 0.011135629378259182], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a48a49cf0677e5ed37b1a5c55213ccaaa3dfdec819ff1af5c7e830558d300d12:action", "state_id": "b04fbf4382bb2e2f591b171b5abbc44d4ccf386cc304ed50d90e0d264e7fda75", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2890625, 3.103515625, -1.328125, -1.140625], "student_probs": [0.004414296709001064, 0.9701266884803772, 0.011539616622030735, 0.013919434510171413], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6998d0afb8e4448d402c9969b5c84dcd6cfb83a30aacc7fc6da7c2340be59a25:action", "state_id": "4bb25b849ebbc9966ce38878ad4d92b6477a25383f1d19e474b5b5ecdb8d0b4f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.171875, 3.1953125, -1.19140625, -0.966796875], "student_probs": [0.004519525449723005, 0.968350887298584, 0.01204772386699915, 0.015081745572388172], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5e2ffbcf5a870321c44f534804ceb2781d6ad8c3bffa55a7d6b45d283e524c71:action", "state_id": "17d354e7097bd08e87d76af757da6c3be393dd57c26af7862b903ec785c340d5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9765625, 3.16796875, -1.025390625, -0.8671875], "student_probs": [0.005614435765892267, 0.9628256559371948, 0.01453432347625494, 0.01702556572854519], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d6723eb7a23bd8c430f54a379a54a229bfbd7abf5ea06e08938e15a7f8cfea9f:action", "state_id": "5a7c5cad7fd30a27cd6d8af58348ddc26f826e03cd9b14e279b177242001ab86", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.72265625, 3.20703125, -0.828125, -0.62762451171875], "student_probs": [0.0069074248895049095, 0.9555474519729614, 0.016896866261959076, 0.020648211240768433], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "740d71f4dcf43166d0e85d7c22775ca08028a0606d5f49d9292169f41ceace73:action", "state_id": "cbf1ad359a35b66a427f2225dc3814b95e9786a4bb6a565434f448d803dd3923", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.74609375, 3.19921875, -0.7705078125, -0.658203125], "student_probs": [0.006796457804739475, 0.9550026059150696, 0.01802910678088665, 0.020171932876110077], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "91ccfd80e266b31021de0ac1614f856dc2ff8517c6f351ecb70dd9e0e54cff56:action", "state_id": "6f6d56f847ea42e9bb9ba097504e53ed62fab27e44ab47dc989fb759b52af203", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.703125, 3.23828125, -0.7216796875, -0.559814453125], "student_probs": [0.0068132709711790085, 0.9536327123641968, 0.018179919570684433, 0.021374164149165154], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ee12cfff8ec52d9d2b519fbb8e990988c1744d7f1e9975cd9702f161bd98e24f:action", "state_id": "df42764658e2e61219d5ab8bc3249acd469ded553e67eac4a31c66556d336339", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.578125, 3.30859375, -0.6495361328125, -0.3828125], "student_probs": [0.00717599131166935, 0.9509482979774475, 0.018161969259381294, 0.023713713511824608], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a56bfed69d2fa4cf0225bdc40dc1848c1eec32aebd674e023b9a54a4ea134cf1:action", "state_id": "211d676ffafbe5703cb2087b79a881ccb47b08bbaf22d275436ad9e859e26773", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.640625, 3.28515625, -0.716796875, -0.54248046875], "student_probs": [0.006929312366992235, 0.9548380970954895, 0.017454344779253006, 0.020778214558959007], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b404370ff60036b078dcd2dda4882099df763db7499fc428971705c5e0f1fc16:action", "state_id": "0175b3db65ac4467f396363dc476bd88ee39fc546c33bfd44e5b26e7921b023c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.63671875, 3.294921875, -0.4609375, -0.244140625], "student_probs": [0.006808620877563953, 0.9437206387519836, 0.02206452004611492, 0.027406156063079834], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0050f61c63a53fbb3e101e6dfb8daa8d43ac925a91f0000f84dd96d4a882e0ba:action", "state_id": "b94f701936e13cd3dddcf79f4f7c0b5e37194014d8927366d9f564f8d7c509b3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.703125, 3.310546875, -0.333984375, -0.14453125], "student_probs": [0.006244524847716093, 0.9395273923873901, 0.024553285911679268, 0.029674818739295006], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e0d8edfa726559279c4843ab1ac3f7b7a2a7b246323015563cb277a8a364ce98:action", "state_id": "e7661e029c6d3053ef5d3556dc0b53bb24a34f126f12fb6252485e3e32866014", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.640625, 3.322265625, -0.310546875, -0.03515625], "student_probs": [0.00654586823657155, 0.9361022710800171, 0.024752147495746613, 0.032599691301584244], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "07f90ce5c664fe0641fb4d1cbcf83b54428ce87fb97960cd0cacc9d44946e607:action", "state_id": "8b54a9e04e02bc8128d3cd9f54236a9b73d115be84fc9f51de7a9e099310e567", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.91796875, 3.134765625, -1.0234375, -0.859375], "student_probs": [0.006143329199403524, 0.9611218571662903, 0.015027743764221668, 0.017707008868455887], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "919e66a1a51648a70312b2f82eae33cc855b0542340c7b4d08dfdf6726393c01:action", "state_id": "09043dae7bc68ff38f396f003f356e9f0998c3d781b8c6b042f6715d971cb2bc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.953125, 3.1328125, -1.203125, -1.03515625], "student_probs": [0.005975403357297182, 0.9664108753204346, 0.012649929150938988, 0.014963596127927303], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a9302f8f295ea67c2d436c1f1866df426f1cf30888739a2d35fd3eaf01948122:action", "state_id": "bccc8a0c49ce7105227278501b830eb762b872d1edcaffe53235029904922bd2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.03515625, 3.126953125, -1.375, -1.109375], "student_probs": [0.005555829033255577, 0.9696711897850037, 0.010751055553555489, 0.014022018760442734], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "280d921c14675bee813fb8d4cdaa326096755f2f76776c44b4f9afc06231d480:action", "state_id": "349661c313688c5ad4045af48c43f32820eb6b32092c9e9494123742360725bd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.95703125, 3.150390625, -1.359375, -1.10546875], "student_probs": [0.005868374370038509, 0.9697124361991882, 0.010667843744158745, 0.013751394115388393], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "40df8f6db2433781fab57817442c9a8c7f6500698f940bb58b31d8c47e644a75:action", "state_id": "58342a407c10ae6a4db2c8f9b1af26320e74b40f65cf60ef8adda38fd2ddcc2b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.92578125, 3.15234375, -1.4140625, -1.12890625], "student_probs": [0.006047424394637346, 0.9704476594924927, 0.01008804701268673, 0.01341679785400629], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e3e1698b6d944371a501676cad1a46c01ff8e66d23c6c82f98ce2917e0b8814b:action", "state_id": "0ae85f722c04bde5e8ab6e51cd0bea678d4a34213a778dd201970ed02c5a114d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8828125, 3.046875, -1.2265625, -0.986328125], "student_probs": [0.006958226207643747, 0.9625750780105591, 0.013412331230938435, 0.01705441251397133], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "94aa0696fdeed36ca4a3c3be97d9d058066105da017ddf2fa1e0a06fe9853ebe:action", "state_id": "289b3bbaffd490785d7ee76f58f2344f66c04a267460bfb3e679a43738e5719d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.94921875, 3.0234375, -1.28515625, -0.9921875], "student_probs": [0.006668596528470516, 0.9630118012428284, 0.01295487117022276, 0.01736472174525261], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0bcf954973d3fbecdf02a5484cc45f24382d2e6d244854c9eb3d05f26c07108f:action", "state_id": "195ff515a711c9688f7671613ddbd8696d4968f2030c5b9d42f1ed1873e89996", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.92578125, 3.056640625, -1.3671875, -1.0390625], "student_probs": [0.0066224075853824615, 0.9657266736030579, 0.01157737523317337, 0.016073593869805336], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fcf728615fdf0fc4c9e9cbed359173672d2eb1cc2fbaecc4262d1221ac99d65f:action", "state_id": "8dd33d57e9c1ca3d909592e91252c3420099364d9a1ebe77cc7aa54a46c1704b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.01171875, 2.876953125, -1.484375, -1.103515625], "student_probs": [0.00724892970174551, 0.9624919295310974, 0.01228277012705803, 0.017976349219679832], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "31a9860b2ea822c2083a9a0a9a48090cbef0dd5e19453fcc608b9c54ec4e2d27:action", "state_id": "984f73a808ef018e7231e868e3a326e50a8ecb84e87cf2d02266473df0f5a23b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0, 2.87890625, -1.359375, -1.033203125], "student_probs": [0.007298526354134083, 0.9596596956253052, 0.013850170187652111, 0.01919153705239296], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3dc3656f6b0701d3efc5727abc3d9f867fd0e55173ea4ce0410cac42839e2794:action", "state_id": "9131b5e3b6b536e85be41b5dd521463cda0fb663e28630a375dff85919a8082d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9609375, 2.833984375, -1.18359375, -0.927734375], "student_probs": [0.007881419733166695, 0.9528237581253052, 0.01714749075472355, 0.022147202864289284], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c080358338b49bdabd72963f4f75ca0caedcca6dccbfc9d3a6f0c0408a85bcaf:action", "state_id": "c44b867610f2e3c20e2ec3a1ee4593ed6c13ded63803662934d0326d68a6311f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.95703125, 2.845703125, -0.859375, -0.7607421875], "student_probs": [0.007743050344288349, 0.9434374570846558, 0.023206952959299088, 0.025612609460949898], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "70b22d137969c00aa5cce9e164e660aab744d4af19e92d2be54d930a3198d0b6:action", "state_id": "a8d2cdce5ce34f0c13d04f4fce548a91eff887c7b50f1743fb0f70428513bad3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.02734375, 2.861328125, -0.622802734375, -0.7158203125], "student_probs": [0.00706401327624917, 0.9379392862319946, 0.028776364400982857, 0.026220375671982765], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7dfdfa90506da67a3c0983959e6918ea9c110b014ecd34056864bd71882ffb29:action", "state_id": "43c14e7c859eb58323ec114d8018cda9c36d7260e760ecb9ee040811b7c2f2a7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.828125, 2.939453125, -0.42578125, -0.146484375], "student_probs": [0.00780804269015789, 0.9184912443161011, 0.031737469136714935, 0.041963279247283936], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9b3b4a4d4ca3186f90a1b9e4b29291fb66c1bf7f4592a47d4019406d3d5ace4f:action", "state_id": "655f488a1bba98acc3b199071ba737c48c82f76fe0ad298a5396950dfe7e18c8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.60546875, 2.9765625, -0.326171875, 0.046875], "student_probs": [0.009300077334046364, 0.9087353944778442, 0.03342551365494728, 0.0485389418900013], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bd91b9a3eb4c1debbeb02ed6295a16d1ebb1ad4676ffc00618d2c985d8d0c2bb:action", "state_id": "d1d55be54e04f940f7245a5d553c8b77f47ebab168f28a283e11b22b40041dab", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4921875, 3.02734375, -0.314453125, -0.03125], "student_probs": [0.009965164586901665, 0.9147279858589172, 0.03235698118805885, 0.042949844151735306], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a6694e08f049a538e4bacaad60401c03dfd69fda7f58cde03836e68c2802cdce:action", "state_id": "8af3c44c8be3f87b515d31b0657ed445eb1d6d5e386032f9c45cb2164fc8bed7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.46875, 3.048828125, -0.310546875, 0.03125], "student_probs": [0.009972142986953259, 0.913582444190979, 0.031753361225128174, 0.04469204321503639], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "faaf9fcfb387a4e174a3a5c59a372772f611a21aba181a6ec8c43c249570f940:action", "state_id": "a949daa26b6e614583794ce263a8b31130c6d9903f1b7b0c922a0d62a50baf25", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8515625, 2.826171875, -0.83984375, -0.4638671875], "student_probs": [0.008674371987581253, 0.9327220916748047, 0.02385733462870121, 0.03474612906575203], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e3c601fd19336182bf04d676ab4f3b73f817a40fc706a358bd3e0f01df0803cb:action", "state_id": "2548b57be507b808172d92a8baa82f616d161b8634ff08491dbd9aa63acab8a5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7109375, 2.888671875, -0.8251953125, -0.2109375], "student_probs": [0.009315156377851963, 0.926349937915802, 0.022587232291698456, 0.04174762964248657], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "90f1eb08ea6316f877e06b698db98fe459b04e21001db658a8586abf34fea8a1:action", "state_id": "6640d2ccdb0e74b4bbcb9b776bd44d859a87c079e5dd12922d5a0cfaef69e82d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0, 0.6640625, -0.47412109375, -0.173828125], "student_probs": [0.0382208377122879, 0.5486404895782471, 0.17578467726707458, 0.23735402524471283], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cbde3ac7890421445d25c05c9da9c2ace64654e5ad93395d7aa8e751a748a3df:action", "state_id": "bfbe034dd4d5dd59c2561b7e5197e5e2a26fc5645c95ed0cafcc7a94ed19b3be", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.828125, 0.3203125, -0.154296875, 0.1171875], "student_probs": [0.04566248878836632, 0.3913939595222473, 0.24349714815616608, 0.3194464445114136], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d18299d69e9592ae3abeeca51dcf233e7d7fa5feef05b24e4751f6342e019e9c:action", "state_id": "473df85577079655547cb6f67321db2e811f94ba30506b6e295e02c6e22db220", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.41015625, -0.220703125, -0.03515625, 0.328125], "student_probs": [0.07179822772741318, 0.23587758839130402, 0.2839674949645996, 0.4083566963672638], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d7d8ae5e123d210b650fec9f5e218274c47a2ef97e29631e732630a0ec9112dc:action", "state_id": "2e6a5f0331fb19f32a57639f3e75af7683eae3e06a899f5c54e56ee2c69494ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.94921875, -4.015625, -0.7900390625, -0.558349609375], "student_probs": [0.27046018838882446, 0.012600274756550789, 0.31712770462036133, 0.39981183409690857], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0d5fd3c581bce525779dabe8d7f57ca95b4e77f816688677d649fc7d69063c5:action", "state_id": "d02f273f58cefefd1ed421af338718bdbef733489b0f00731a0b1ad082e8faab", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7880859375, -3.53125, -0.7333984375, -0.556640625], "student_probs": [0.29577070474624634, 0.01903768628835678, 0.3123961389064789, 0.372795432806015], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5dacc8315cfdca5a1d369fccbb95f60c45b895e7aa1111f0ecd39dc57314f823:action", "state_id": "0bb33e4eee4e9ed102b251e97efa6ce59ca1bf7f5a9e442c8c71f3827522a56c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.68359375, -0.419921875, 1.65234375, 0.38671875], "student_probs": [0.21233727037906647, 0.0704328790307045, 0.559434175491333, 0.15779565274715424], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d90c437a27566de7317f6792abef8f975405a15be50d682b65f94bb7e408d569:action", "state_id": "2c5469fb36a24249d597e636e3ca431b22f6b067e393a2b7475880e0d77a1479", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.927734375, -0.19921875, -0.234375, 1.0], "student_probs": [0.36876076459884644, 0.11948549002408981, 0.11535780131816864, 0.3963959813117981], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0fe0e64ff9394d073950bd104d811d98017877fc499b2211243f237df84db2a5:action", "state_id": "7c611f9950c516eb7fdbf9f06ce5924aab295a18b1415dd297ba40495bce1180", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.48828125, -0.427734375, -0.54296875, 0.5546875], "student_probs": [0.17103292047977448, 0.18170835077762604, 0.16193071007728577, 0.48532792925834656], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "52a8a26a2ddefd2316a9665603dd8275a8da15d80c368e40b15deaefe4f04382:action", "state_id": "350d1433c5f38520752682e686b0e43dd310947d43a2e604c42349eb466727f6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.75390625, -2.97265625, -0.91015625, -0.2578125], "student_probs": [0.2772860825061798, 0.030153464525938034, 0.2371753603219986, 0.45538514852523804], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c83bae76b26d19888a55fda92d34433d8e6b2cde9158b3ab38ffd389b373006a:action", "state_id": "dd17d61fe23cd47fc0576db359e69c29d70889c14f068c56bec79f9c0e218484", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.27734375, -2.765625, -1.0234375, -0.6136837005615234], "student_probs": [0.22438137233257294, 0.05065641924738884, 0.2892390191555023, 0.435723215341568], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "096d9e85cf1884dd5d7f7c7668ea802a9d950421cc0236d81c3a965958389a4c:action", "state_id": "f4869ad46e77a950bd12b314aba9421b22403adb1d82212c62f293a3cc9009ca", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.796875, 2.484375, -1.19140625, -1.609375], "student_probs": [0.004857326857745647, 0.9550257325172424, 0.024190427735447884, 0.015926562249660492], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "535db078b2f6b7b47b4a850fdcbeb0cce0df8bb38890be9b23a73becb35c750a:action", "state_id": "6cb9fbb7fd49d73c39df2ad9eabed9ec5d78c03f9855b053e24279e0177ed9cb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8515625, 0.30859375, -0.692138671875, -0.23828125], "student_probs": [0.05592890828847885, 0.4850430488586426, 0.1783067286014557, 0.28072139620780945], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ad3caee9c76de619b98dc1ec2f159369d42d945fc957dd04774c5ecf8fd42e0c:action", "state_id": "730d381a6935764d2f949f503a6195676ad651d3187bb11a7f32cd47b140573a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.84765625, 1.05078125, -0.791015625, -0.34765625], "student_probs": [0.03772994130849838, 0.6846388578414917, 0.10853737592697144, 0.16909386217594147], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "215de9de06178f3ca2dd795cd251496cb4b3f4d9a94e64542d7c57a97799345d:action", "state_id": "fcc07892241a9cde2ba462cb9f88c289f596eee1e4d977ce48b08de7ffb74f2e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.64453125, 1.548828125, -0.3896484375, -0.09375], "student_probs": [0.029768338426947594, 0.7254591584205627, 0.10441029071807861, 0.14036226272583008], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e512e7dae82e37a150043d9d087d8591c0c71bcdb90f026bb1ca5f54026a237e:action", "state_id": "d5a5ea0c1edf6c56d704ff164310ed81ef5b535c80fb8a2694edb9a8fc54a23f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.86328125, 3.064453125, -0.96484375, -0.58477783203125], "student_probs": [0.006891163066029549, 0.951437771320343, 0.01692306436598301, 0.02474796585738659], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "eeadc1e735b27d9b566f9a0cffed010dea0d0ec28757dc697c424e3868d460ea:action", "state_id": "c84c35b0075ac207a7dc4e98fd92a1beeb76e4705def77d0712ea7e51a1b1bf5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.80078125, 3.05859375, -0.943359375, -0.65478515625], "student_probs": [0.007383002899587154, 0.9519909024238586, 0.017402298748493195, 0.02322377637028694], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2ca9bcadbd5d3accc64d5557d386c97c12c386539c4eaba1109f0ab76318ef23:action", "state_id": "859cc0b5796b0232ff86070a20dadfcf3f68a436414f6d0d5b66854176fa56c0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.69921875, -4.390625, -1.21875, -1.4453125], "student_probs": [0.11009251326322556, 0.020285671576857567, 0.4838571846485138, 0.38576459884643555], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "84004405b8f9ddd4ba8d00baa3c5f9e8360c7a0a4a89f39042e6747537c3a698:action", "state_id": "659895da3b40f166c8309d32ec985a280f6e01c5ef392b64ff5d867a59646535", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1484375, -3.296875, -1.05078125, -1.123046875], "student_probs": [0.14079649746418, 0.04465106129646301, 0.4219858646392822, 0.39256662130355835], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "957092ffca7acab56ca13dc7d2e77ae16e88b4560a7ad47571153fb122208adc:action", "state_id": "2175d9ae9e81ff14656cea1919de0fb318e7ecb2d50ddc706585a7addb9ce956", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78515625, 2.60546875, -0.087890625, -0.5653076171875], "student_probs": [0.011045267805457115, 0.8912518620491028, 0.06029611453413963, 0.03740673512220383], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "970de74fcf1435f3c82a69936880190d76e5a53d8b011c9231cebb0ab03e530b:action", "state_id": "bd7e6b931a4a5b9e567dabc72e331fe6eb2e6256467525dfa161e1205e9a04f1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9228515625, 0.37890625, -0.05078125, 0.33984375], "student_probs": [0.09431696683168411, 0.3466857969760895, 0.22559276223182678, 0.3334044814109802], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "94c325d8e6a1f73b1bda4098ff3271964998d30e9efb09c28e0842a4857f4895:action", "state_id": "5e84a187cb087076f6e9240ee4dd9713ce1446bdbf65d6d07c1d73049f957dbd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.916015625, 1.1875, -0.01953125, 0.44140625], "student_probs": [0.06438295543193817, 0.5276137590408325, 0.15780076384544373, 0.2502025365829468], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3ee8fb1687e5fccf52655f26623cd412bcba142395a1ff678f17ae76b0501f16:action", "state_id": "954ad9be07fa5286709842bc7331c1a7dbb6fc8239a025a917d90522599c2677", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.96875, 1.5390625, 0.08984375, 0.46875], "student_probs": [0.049090560525655746, 0.6027359962463379, 0.14149445295333862, 0.20667898654937744], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "216d28c00644ca127af97599e0533654b749aed73e2bdc1aa0ccc29fd938fca1:action", "state_id": "ea256c6b1257e687423951d8f51ade0bcca967b4efce2c80ef5c9c9facc1676a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.53515625, 3.107421875, -0.552734375, -0.3203125], "student_probs": [0.009021010249853134, 0.9364858865737915, 0.024094369262456894, 0.03039870597422123], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4210f6d06210e565f7d355ea700a87d3ade0ee8b39a41a303c7e3bb594f2a955:action", "state_id": "0ceb7d8b32784ed881c858af4601a8aa23ac2c548656c39b8da0d40a445ac35b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.96484375, 3.0234375, -1.12109375, -0.8583984375], "student_probs": [0.0065345484763383865, 0.9585143327713013, 0.015193280763924122, 0.01975780539214611], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ffa7ae8dd71dcffab6378c167e47751e3003d2d52a03850e3e2e3495d4ea2780:action", "state_id": "d104af853b77581ccbb7d72aefc56ade1dba7e2a3c7e52f72d228139d317e25b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.14453125, 3.03515625, -1.3671875, -1.19921875], "student_probs": [0.005453258752822876, 0.9686475396156311, 0.011864574626088142, 0.014034600928425789], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a5675db2ab37c25b71694e58a9ab1a35e14515d5b9ccfca95d1cb052c08692fa:action", "state_id": "f887e4275047c78d1cb45f96e454dbbb22d027fb9e74bb0b5ae3602d68449848", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.328125, 3.048828125, -1.5546875, -1.44921875], "student_probs": [0.004505773074924946, 0.9748782515525818, 0.009764926508069038, 0.010851091705262661], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b197f3f529d0dab9a4c594bf681a546c02b9d401c5c705303b1217ee2cb32f72:action", "state_id": "57878ae584f6f0779c2a1449dea510659e98c2fb16f2e71ce78e0fc35ec0f75a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.33203125, 3.05078125, -1.46484375, -1.453125], "student_probs": [0.004475835245102644, 0.9740917086601257, 0.010653414763510227, 0.010778994299471378], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4acaa6c953b01565e4144448d8b7fea212ac13f527ae8300be7a9225ef6b54c2:action", "state_id": "6fe3a0414f58a1d53470127dd161a5256cc710e5440d533f19ed44e7550a9e76", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.37890625, 3.126953125, -1.5390625, -1.515625], "student_probs": [0.003971140831708908, 0.9774163961410522, 0.009197182022035122, 0.009415287524461746], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "10e0c29b79aae39d49ac7160ff7a409b56e01d66689c706db110464feed7b3e8:action", "state_id": "8bc9362fffbf52abf53e54e048b33a22c0e3c109d3c0df2661b7e2675acd5747", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.4609375, 3.0703125, -1.671875, -1.59375], "student_probs": [0.0038753554690629244, 0.9783695340156555, 0.008530943654477596, 0.009224149398505688], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c5c986e2bb3f7f6f37dacc6d0de9468ca173f7d870446c9a413928fca50144ca:action", "state_id": "b6f2bf0243420ae1bb1cbe2e36774a52a8b2aaa37c56d3c1aec662a51bca6150", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.4296875, 3.05859375, -1.72265625, -1.63671875], "student_probs": [0.0040472871623933315, 0.9788007140159607, 0.008207743056118488, 0.00894429162144661], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7224b5dd0dc2b704d13d9f424ef728cc43994000071512a1031c44aced28bf6c:action", "state_id": "feb5ee9386cc4dd2b2fb01aa5d4c46441f74aae584dfbfa7276d14b171399e76", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9140625, 3.107421875, -1.64453125, -1.43359375], "student_probs": [0.006428288761526346, 0.9747613668441772, 0.00841688271611929, 0.010393464006483555], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "09566025dc4d672c16f35d458283eda9b1b7833c37a05751794efe4599a068d8:action", "state_id": "cdf4af3e7c92bd303962a4685dfa3f4618fac5ca20a37892ec48e5f20a02f120", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.05078125, 3.158203125, -1.73046875, -1.54296875], "student_probs": [0.005349098704755306, 0.9783939719200134, 0.0073686945252120495, 0.008888342417776585], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5b1061b1a9ec26e2a67f4a0cab84e686ffe81123867e158e5577b2e641bbd8e0:action", "state_id": "8ad4b0cdc289903da819da5365e1a0dede20bab620d3a8223dcaef5c9db51d2c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.07421875, 3.203125, -1.63671875, -1.45703125], "student_probs": [0.004993719980120659, 0.9780148863792419, 0.007734424900263548, 0.009256893768906593], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7d66e3684ddeed95285b34f3584e0eb4a268b5716edba4dc95ee1052c922d8dd:action", "state_id": "4a52cfa4f6571a545586ed38a4ce16ea00ba6caa36c7dd6c606766f810d83099", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78125, 3.111328125, -1.40625, -1.32421875], "student_probs": [0.007281668949872255, 0.9706230759620667, 0.010594765655696392, 0.011500509455800056], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ec2be3f28577d09fe0d88a58c3a75b3eecdf39bb6fa5433955ecc19fb3fb02dd:action", "state_id": "6b424a63e74ddf91bad2e4f268ff072f05c44e7a9d073158a264603943a842bf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.49609375, 3.298828125, -0.9306640625, -0.6727294921875], "student_probs": [0.007940712384879589, 0.9599919319152832, 0.013977273367345333, 0.0180901437997818], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a40dca95a6eabada554a88d48aec5114014b0574bbaf010d9f1a635198284b4d:action", "state_id": "bea4b7289ddd0b0ec7139fd318fcbf191fb27ee81091b0b32fe26ab631376ec0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0, 3.4619140625, -0.77099609375, -0.4140625], "student_probs": [0.011024478822946548, 0.9553064703941345, 0.013861594721674919, 0.019807400181889534], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "46c72207503c68acccd42ec7facf887a3bad466b7735ebc96757f2eccb6799d4:action", "state_id": "88bb1d80ba6492f9ffd22962c7ca48328c15f90d8c3a7221dc2253f4a196b958", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.0078125, 4.09326171875, 0.08203125, 0.3203125], "student_probs": [0.01565251313149929, 0.9454922676086426, 0.017123902216553688, 0.021731361746788025], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2249d6ad8f89c9002a9194996b260122cf3a72352476ebc1f3143499c002fd97:action", "state_id": "44f574a1b5dd7fa034c407868e517cd02a21371a87bbb47eae80ec0e982f1a13", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.45703125, 3.279296875, -1.013671875, -0.8740234375], "student_probs": [0.008448505774140358, 0.9632545113563538, 0.013162198476493359, 0.015134809538722038], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7239bf98901938ca4c3ee1b8301bce8f57cc2af1f37caae6f2d32a77da7911e0:action", "state_id": "026e07d0707a4f980932e2e997d99ff363fdfa504538d55dcf70eac9b2e282e8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8125, 3.203125, -1.1640625, -1.21875], "student_probs": [0.0064319586381316185, 0.969619870185852, 0.01230144314467907, 0.011646772734820843], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5649f9cae1aae80616b57defd43d32476070b6722420274471d89c3acba6ae8c:action", "state_id": "8ca43f0e92bc4860cafb1002df44bcd49fe14146d790348244310f57db997d1e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.96875, 3.0859375, -1.40625, -1.2578125], "student_probs": [0.006190168205648661, 0.9703431725502014, 0.010864083655178547, 0.012602558359503746], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "62cf03902b579c7a5ef1bccad33b20070bc09a66ed1419f2166fafca3d47f010:action", "state_id": "deeec24134556c49acf4f2714b7f6a91b6176b36819ad458ae382351bed24309", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9765625, 3.19140625, -1.3125, -1.12890625], "student_probs": [0.005529910791665316, 0.9708194136619568, 0.010742783546447754, 0.012907750904560089], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7d6ae7a7dec2c92cacf5126c70f1bb609659a061b876215e6c62a3fd066fabf2:action", "state_id": "8f874c6e509e6f739a1185d19e923bcc20dd76bbe85c4e89767a7b457c664eec", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9453125, 3.220703125, -1.23046875, -0.953125], "student_probs": [0.005526195280253887, 0.9682741165161133, 0.01129481103271246, 0.01490485668182373], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0b4a4b8e7aef00f784b3e840d975533ef3d07aa918c3d83c5c884d9e7dce9dfd:action", "state_id": "5d8fade80233b5ecf4d50f118eada950ed9c6904c3a37e4483ed95147aa6049f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8828125, 3.1328125, -1.091796875, -0.63134765625], "student_probs": [0.006351165473461151, 0.9574402570724487, 0.01400835532695055, 0.022200241684913635], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "10ada4e33de70edf6940b8c67af31573eab7c720d3ded4940b40a671fbfbb3fb:action", "state_id": "77840b427a5c18c96487510f5247a1bc70a1773a973eae521350d5b61f7b8bf9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9375, 3.115234375, -1.0859375, -0.7734375], "student_probs": [0.006135123316198587, 0.9598380327224731, 0.01437646709382534, 0.019650302827358246], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6184ab3accf7d7dca6fc7b6eb68318b2ddfcbd531be2211b09afd2b226ecc7c9:action", "state_id": "8177e7d4c6785fa55e3a37643a49039725cd76b9fffb7e4a8a925c4eb0c522c7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.84765625, 3.123046875, -0.787109375, -0.51611328125], "student_probs": [0.006587483920156956, 0.9494421482086182, 0.019024323672056198, 0.024946024641394615], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "704c4a85e32caddcee72f245de807329210defa0fe3c1b21fe8ab8ca84ff0168:action", "state_id": "4ef20895104f0c4d751c10288b38e55019ae56522d1f989a6b80d69750040311", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.90625, 3.107421875, -0.6787109375, -0.29296875], "student_probs": [0.006254368461668491, 0.9410083889961243, 0.02134503796696663, 0.031392261385917664], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "106d7b76ef02c07bfae80bb336644512c616bea6b601f440010c692a866a0779:action", "state_id": "a10320d2677c43be451f3a4e143507a64307293fe6b9f03806c93cf83fe5c021", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.71484375, 3.1328125, -0.4462890625, -0.09375], "student_probs": [0.007296305149793625, 0.9298509955406189, 0.025943543761968613, 0.03690923750400543], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f11cbce76ea73e6d55b8f4316f6f80180f1cd5e02878caaebc269af8fa4a6855:action", "state_id": "ebadc511ebb969d1cfebd3406e5163f142bf23a5b3e55dc58ab0666d5dd264bb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4375, 3.103515625, -0.0859375, 0.296875], "student_probs": [0.009586363099515438, 0.8990666270256042, 0.037036504596471786, 0.05431044101715088], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0eee3a1586a743995c4ca910c24b3adf4c572ffe708d4c3318e4a8c717688a3:action", "state_id": "58adc7863ea2f1b889bb3ef7889cc2e3bf368f7b12867109d1df95260ffca926", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.734375, 2.98046875, -0.482421875, -0.111328125], "student_probs": [0.008253748528659344, 0.9210472106933594, 0.02886473760008812, 0.041834209114313126], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a7efe5186d8fab34c76641fc332f67d6feb6213a5e0d5d1565244247ac83a49e:action", "state_id": "b050e552ebbe9a1a8d2a495b7f8ca11de68482b1cbecc4c0f7472f3819d28972", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78125, 3.01953125, -0.55126953125, -0.29296875], "student_probs": [0.007665419019758701, 0.9321562051773071, 0.026224644854664803, 0.03395378962159157], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "97013790d2e099d25077cab8870cdb837c42940edf211dc6607a507fc2cdc6b8:action", "state_id": "07259e39296092e64e78814a32707d9ec5d72a5bc4f5188c083d322dfc6e41fe", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78125, 2.861328125, -0.534912109375, -0.07421875], "student_probs": [0.008787207305431366, 0.9122143983840942, 0.030558252707123756, 0.048440106213092804], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "65f9f2857f37b42e1060427728357fca9109d5b44207313e6cd52cd5b9c0474b:action", "state_id": "e595b1839487b3f7d2392ea3fdf8f7e7113ad950c15461af02e271fc8fc5002f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8359375, 2.841796875, -0.2421875, 0.05078125], "student_probs": [0.008330137468874454, 0.895707905292511, 0.04100237414240837, 0.05495962128043175], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dee8da9a879b4430cd0a15e1386d1e7cde31119c070e15b790a3dd152ba99532:action", "state_id": "5ce15de56b536b1eb2984a2137a1c3ce0c846d3a8d456c964ee8b8ca25e308d9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8671875, 0.46484375, -0.00390625, 0.40234375], "student_probs": [0.0364716537296772, 0.3756157159805298, 0.23505431413650513, 0.352858304977417], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bf3bcc1ca81166ecd33f2c826f85e73c94bc2dc2d2b9301448c00f86a8d8469c:action", "state_id": "5b24fecde0f43ea387c08892aef8fd6b6e15f0e95dc422603a563a144531e1a7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6328125, -0.09375, 0.22265625, 0.62109375], "student_probs": [0.046339668333530426, 0.2159530073404312, 0.29632803797721863, 0.4413793087005615], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bf9c8063a417cdacf495257971fa2395de5041494928ac65b2fa58c32218e5fe:action", "state_id": "157cefa53093c3bf75c5125cf2c265666e171332e44b095d1df5c779e8168dc0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.51953125, -0.4921875, 0.24609375, 0.546875], "student_probs": [0.05702797323465347, 0.15931536257266998, 0.3333413004875183, 0.45031535625457764], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "50abe8c5c0de8c08f99cf421a685b27184fdfdfc22c49c6266f9f96e780b1bfb:action", "state_id": "970e09f4433ee8c661a44eaed7cd9533fcbc7a9667b846c05c7cb2100bd2415a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1015625, -4.453125, -0.97265625, -0.728515625], "student_probs": [0.27587994933128357, 0.009663957171142101, 0.3138364553451538, 0.4006197154521942], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dad1912cd85de2adb4b1461c91071af10ecd03c0524d7fca35a0fc67194ca311:action", "state_id": "0a18aeec2895f857059c0be2290a3115603ec38fbea1d0bc04114bbf77500dda", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.001953125, -4.0078125, -1.078125, -0.8427734375], "student_probs": [0.31759122014045715, 0.01571955904364586, 0.29429811239242554, 0.37239113450050354], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3f7b7861d331e34bbfbd3e104afc483b9ede1404aadf32da00af34aa637e5348:action", "state_id": "f4d5eb5b78e85626503dcdb9814cb9c2d7995a2b14bdc7ba8e3fae0575fec05a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.828125, 2.615234375, -0.0546875, -0.571533203125], "student_probs": [0.010475074872374535, 0.8910120725631714, 0.06170939281582832, 0.03680340573191643], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8ac982c30b1089e057b39c4c5ca7b19c7e4a7c54d74efae13db1051d2ed3a0eb:action", "state_id": "0d56710e9464d14a856d6d75b50d7630569d2919fab116f400514f256ff38d2a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0341796875, 0.31640625, -0.1015625, 0.2734375], "student_probs": [0.09010478109121323, 0.3477762043476105, 0.22896985709667206, 0.3331491947174072], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a5c6a98ea5fc831cdf842d2b266ded87442b346acaaaa6629e0f0c38fc0f6660:action", "state_id": "73b48cec7a69c113579c089ef9ffcd031e2025e3d14cb154d800a1f03e1052aa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.73046875, -4.078125, -1.1796875, -1.2890625], "student_probs": [0.09802348911762238, 0.025471264496445656, 0.4621956944465637, 0.4143095314502716], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0c2420b3a0ebc0fd6e26b8949c778a4521bb4060613b4b70451847010e324ab:action", "state_id": "0346985157d5ca4440c1e3e419ea8422e79912391ffbfb113d2b2bf46a211569", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.76171875, -3.515625, -0.9765625, -0.791015625], "student_probs": [0.16651000082492828, 0.028822291642427444, 0.3651147186756134, 0.43955305218696594], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "93d9a870c0b3d78f62859c30f707cfb87c9201693007b65eb3c34c4243348c5a:action", "state_id": "51b3a217afb614d203e46f7f5f31173ec2cc2492739dc4fa60684fe430cfcd2a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.71484375, 2.48828125, -1.20703125, -1.578125], "student_probs": [0.00525008887052536, 0.9546741247177124, 0.02371380850672722, 0.016362035647034645], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "40782d667576971b53e40229d94f3a261edb73f4a473f2158966859709d7e4f1:action", "state_id": "883fa2e2d1a465e8f0dcb540cbc894b888a305b562a3748ebb669ded6c05531b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.984375, 0.421875, -0.966796875, -0.421875], "student_probs": [0.05094371363520622, 0.5650822520256042, 0.14093509316444397, 0.2430388629436493], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "04d233178386af8a7004a98a716f794ca8ba09ea56bd9d58bd033769804b74f4:action", "state_id": "d988db1159080a802ee12efb34b8cacce2da2dbf25d6211020042b1977990c1d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9296875, 1.28515625, -1.08984375, -0.5458984375], "student_probs": [0.03105069510638714, 0.7731437087059021, 0.07191357016563416, 0.12389200180768967], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f38311c313876e1cb76a598987780957135e43d8d5c78a1fe855c89795b81ed1:action", "state_id": "61d66450e8a1f8e7a069db75e16e26a9818744867dd960417df92608609b2821", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.82421875, 1.697265625, -0.8154296875, -0.3779296875], "student_probs": [0.023909593001008034, 0.8089718818664551, 0.06556675583124161, 0.10155178606510162], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0bad5f08232411cf176e87780a8e8dc06ea72581133cabc1186f425d7b8bc8d7:action", "state_id": "0b2155b954f26e1919160ad33073495e42066cfe0bc9c7bb4d21a3fb9455244b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.02734375, 3.04296875, -1.23046875, -1.10546875], "student_probs": [0.006062198430299759, 0.9652479290962219, 0.013449574820697308, 0.015240364708006382], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "daa6af4926ed0c742f2771b6988cea619130d919499b210c101b2df3b48bada2:action", "state_id": "a3972c144831b97ac5b15b99974cef751cc7e6887a1c35b38590c70fe6bf169c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.171875, 3.0625, -1.40625, -1.30859375], "student_probs": [0.005177776794880629, 0.9714121222496033, 0.011133970692753792, 0.01227613352239132], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7a1f3f8095a9d6de4948af614687b3e9ed122f66a466a3725800d7679d56c312:action", "state_id": "dff9033c4518b68e657ae7f1a1232520e538e8f81d4d495154923a34c9b5ec37", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.296875, 3.06640625, -1.6328125, -1.49609375], "student_probs": [0.004574690479785204, 0.9763491153717041, 0.008887105621397495, 0.0101891178637743], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "69777e61048821339399240f0af7993d2a81dd1ea594021e88c377c9189d5af9:action", "state_id": "6364d05632b92ec99b4f3b3315083e89927acbcc33948869dab4dda18e8ac4bf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3359375, 3.0625, -1.61328125, -1.40625], "student_probs": [0.004412004724144936, 0.9753209948539734, 0.009088277816772461, 0.011178772896528244], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "087ef9b7f6220f2c41c5d5064c23fc8809da2ff1bf90e6590fbf0b184093eea9:action", "state_id": "5058e7e5c8d8618148135d5ddbd9e2101c38853ab3f28193e334586bfbbc1996", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2421875, 3.0703125, -1.44140625, -1.14453125], "student_probs": [0.004782831761986017, 0.9702296853065491, 0.010652707889676094, 0.01433478482067585], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "124fdbb7c7d5df6384bbba5ad5e02f0c809526bce94a0944cba2c695dc5f4204:action", "state_id": "f488208ed77bf80f74f2f6860066dbc6dd12e0d337c25b1e45c1e7970a40f10c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.24609375, 3.134765625, -1.2734375, -1.001953125], "student_probs": [0.004457842092961073, 0.9682828187942505, 0.011790817603468895, 0.015468496829271317], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "87fe4bc8039f173feb0e559db86782ecea809863d88acf98cba8846775a1aed4:action", "state_id": "ec071ace1ef53ac6a262b49f832e0669b206a00c4e1c451acfccbd43e0c31f20", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2109375, 3.091796875, -1.2578125, -0.95703125], "student_probs": [0.004808082245290279, 0.9658732414245605, 0.012471215799450874, 0.016847537830471992], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "287dcd02e06ba608155fbfd266df20590a38dc2a6710a34714e3bdd2f9bfd02a:action", "state_id": "078a635832dc915c46f47a597b8e75c8b21163dcb708c6a7329ec1fd8ed6d2cc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.015625, 3.08203125, -1.087890625, -0.8271484375], "student_probs": [0.005866865627467632, 0.9600417017936707, 0.014835973270237446, 0.019255505874753], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bd90631c45000b77841081411e5c891882429346f0514b357551d46e36c6d575:action", "state_id": "aebf487b0f7341f25bb442255c9f0313168ded795719c9b16de4a55689774f11", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7421875, 3.185546875, -0.7978515625, -0.68212890625], "student_probs": [0.006919265259057283, 0.9553177356719971, 0.017790161073207855, 0.01997273787856102], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81ea9adaedc860c833030d982551d96421c1ab0b297dba454db76d026acb4029:action", "state_id": "410980259c52c5b616511a47f66131293c4e5a8b5e7dac7d657e73a18d701535", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.80078125, 3.1796875, -0.8740234375, -0.775390625], "student_probs": [0.006585139315575361, 0.9584181904792786, 0.016636069864034653, 0.018360581248998642], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8b23c19d3da4f75647956dc97cde36f04e12cb422ff50e1d5d8172904ff9517c:action", "state_id": "d262a0ad203b541c1c2007d21320f9fcc8d67cfce7c96c9d581d6c40e0b5cc42", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.74609375, 3.216796875, -0.8779296875, -0.68603515625], "student_probs": [0.006699015852063894, 0.9580033421516418, 0.01596062444150448, 0.01933697797358036], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81273e14bbddfa205435e9cf0730d5851787dedb8ed995647bf492b2d6b46af3:action", "state_id": "a605ebe9c229b84b946e42d3a052b7e148f2f783eac505ce4318ac32a7f83a04", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0078125, 3.107421875, -1.05078125, -0.98046875], "student_probs": [0.005782439839094877, 0.9630063772201538, 0.015057209879159927, 0.016154026612639427], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6b1591131e2550f279005862f86a08cfc9d0f91222ba43e5821e42c56fb7f58b:action", "state_id": "53c2c73bfbb33f72e1d065fbe236ff5da967d6dd6c41fd84c1fd566982903648", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.96875, 3.14453125, -0.947265625, -0.87890625], "student_probs": [0.0057814838364720345, 0.9609684944152832, 0.016056997701525688, 0.017193030565977097], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8a7205f09bee5863cdcc56c429e1db16d436b9d65d4c0b850736924d86ff856f:action", "state_id": "9b7b29856e029fc1509be067a6e85fd80d29e2c8c1be58024ba82ef7477c83b1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9375, 3.18359375, -0.865234375, -0.59185791015625], "student_probs": [0.005705120507627726, 0.955713152885437, 0.016670316457748413, 0.02191138081252575], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0270f682cb3f60c46d1d0df1aeb6051ac0ee95d84253e4396dce954053e013a7:action", "state_id": "5d4b6bc563a702e2c68862142368122de05298a379cd6b7229ea101eb78afa55", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.08984375, 1.71484375, -0.7158203125, -0.5562744140625], "student_probs": [0.01834959350526333, 0.824102520942688, 0.0725032165646553, 0.08504468202590942], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "512f745dc67a50e7e0c8d146746b5e79a385d411169ff5062a0a6971feba1489:action", "state_id": "62b653b62e748e0eb791bb4176b115ccc2e39bfe504f62d9285d8181c8803042", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78125, 1.126953125, -0.36328125, -0.0859375], "student_probs": [0.03460097685456276, 0.6340228915214539, 0.14285793900489807, 0.18851818144321442], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "65794ed41c5d497d9ca26f7d55cd3426fb345e6d366bcf5725f36826b7d48a43:action", "state_id": "d2ccdbdf928476528c53266d1a1b528fadb7ece3f498a06e1bb6baaf9229f1a9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.52734375, 0.7734375, -0.455078125, -0.09375], "student_probs": [0.05525560304522514, 0.5515601634979248, 0.16145643591880798, 0.23172780871391296], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81fec5eba5a11851dd735f5d519d716254181a82608fa0fca01462d9a849b492:action", "state_id": "e2d0548c63d90d105adf41a0ffc95d5ed8d3286b5c1aefa6511e116c226e56d3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.87890625, -4.1328125, -1.017578125, -0.90234375], "student_probs": [0.16322261095046997, 0.01713646575808525, 0.3862338960170746, 0.43340709805488586], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3440fa2b4d0073459a247fa5d04d550f8540b65f878d43c8c62cc18914b49157:action", "state_id": "c87a557a029d8d385f334878baf1fd8b4bb26a786a373dd1447cb02e2d5f6c82", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9140625, -3.27734375, -0.85546875, -0.645751953125], "student_probs": [0.28883299231529236, 0.027182335034012794, 0.30626243352890015, 0.37772226333618164], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "34fed9543ee55a805c70b69b738af69c2777f75124752af2d75fe8e5b5cf3b26:action", "state_id": "89dc29162b95b64538d13c3eee2f1677cf4f31d1a48dde4c8ccd8296df248eae", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.265625, -0.205078125, 1.515625, -0.5244140625], "student_probs": [0.17957407236099243, 0.11215531826019287, 0.6267750859260559, 0.08149556815624237], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a4e858c5863eda1170d10b2a3162b9bb4e22804975da569008961497a2b6ac0e:action", "state_id": "735c2d682b4907550471749d1efc3664165acce467fdfe30360e7c710a43ceee", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.39453125, 0.4296875, -0.015625, 0.671875], "student_probs": [0.24882133305072784, 0.2577245533466339, 0.16510453820228577, 0.32834959030151367], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7c40c98f88473717fe6d47b4892df162315c48bf04ff1b8669541a99cdb114c5:action", "state_id": "2d6d330d5d9bc588fdfc1b907119cd763f6adce29266f4078fda508d7bb266a5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.32421875, -2.7890625, -0.8447265625, -0.333984375], "student_probs": [0.37459880113601685, 0.03184918686747551, 0.22259362041950226, 0.3709584176540375], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "35a9e90740f4f397839d9f7200bcb82e74c686a44730e1439140b7246f504253:action", "state_id": "b77b1e15ec234cd3a0b1e44b0c774dcffd85c0d6baeedafab044b2fbb7a00160", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.400390625, -2.07421875, -0.3564453125, -0.4375], "student_probs": [0.3128888010978699, 0.05867535248398781, 0.32694539427757263, 0.3014904856681824], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "78304dabffc84efc50301e619dd56a4968e7ddfbec12ad0f27e06d05c3363380:action", "state_id": "39c66e989ea2023a6721a9a36fb08df7d46a4edc50afbda427143bad8ea8b677", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.7265625, -2.7265625, -0.6123046875, -1.80859375], "student_probs": [0.07820054888725281, 0.07820054888725281, 0.6477692127227783, 0.19582970440387726], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "01f09ad6539595b98ee1c3305552c34be903ab5a910e4e2ef6bbe055d06536ee:action", "state_id": "bc63555edb052cbc2fa71123c1e634e359f64b6312481ba3c82e0309e2efb0ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.736572265625, -0.06640625, 0.06640625, 0.36328125], "student_probs": [0.12209314852952957, 0.23863862454891205, 0.2725338339805603, 0.3667343854904175], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3c19fb6eea194e2d511f739e345594bd7329ca90ab01b37552a5c2a9dcdc3efd:action", "state_id": "5d05e6bde175d31a55a7ad38f5aad7fd556ebc774371c434e03e69472bb421f2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.61328125, -4.078125, -1.0234375, -0.8388671875], "student_probs": [0.19770467281341553, 0.01680927164852619, 0.35660120844841003, 0.4288848340511322], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b22a00abc61db73ccc656351c9c8e00cbec7a3a557c0a106499e201b5dec78d9:action", "state_id": "d85af456736f6e23bea859ec12676fb83081f62484516fa02c4bd84b9991c88f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.943359375, -3.8984375, -0.90625, -0.734375], "student_probs": [0.30099570751190186, 0.015674227848649025, 0.3123752772808075, 0.37095484137535095], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a5a0022d4d1c092383c613a754745f68f537f6cb5b0ad57d0ee9c83f45a6ad8a:action", "state_id": "2749ddcfccf9fd1158e9cc853a9af395679aca6a227fdf06cc2354de8fa630e8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.783203125, -0.12890625, 1.7353515625, 0.423828125], "student_probs": [0.21317146718502045, 0.08562587946653366, 0.5523850321769714, 0.14881765842437744], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a8f570f9814287f27cc9bc7d7929d643c18a340536c9b4649439cf66c08dfed4:action", "state_id": "4b20ac195e0b7730e3909d2f2b964f6e44ad1f92952d03c70bb629dc491510ad", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.65625, -0.03125, 1.5625, 0.59765625], "student_probs": [0.20321299135684967, 0.10218191146850586, 0.5029569864273071, 0.19164809584617615], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d33c0841f8a10acc25981edc76e5accd1e52fb78d6f9073c553a0a213235460d:action", "state_id": "a8e8f21dd1047772417b1e3534f601d82b815bde49413c3bcbe7f931521ddcfb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.244140625, -2.53125, 0.203125, -0.1953125], "student_probs": [0.26913341879844666, 0.027333088219165802, 0.42093268036842346, 0.2826008200645447], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "16cd8383344ced68491c609c42a55b247f8afe5f8f9b9c48e2dbf2a152b50bb9:action", "state_id": "93ceb617af3fb5afd6ef0098339c4b21bae6c5e3a5698ca7849ce09844febe02", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.08984375, -2.12109375, 0.25, -0.109375], "student_probs": [0.28436896204948425, 0.037301093339920044, 0.3994610905647278, 0.27886876463890076], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "502da2b6e2127aac6e00ff99449a9847dbaa9b895a20adf444318ac926d7016e:action", "state_id": "0989399fd2756a52493f921e11c2f18df5b291de8fc339ddb10e3faf80d4d047", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.30859375, 2.158203125, -1.96875, -1.421875], "student_probs": [0.029032936319708824, 0.930040180683136, 0.015003367327153683, 0.02592349424958229], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "42dd9754f61acc5c62eba52899c7a99685f47ed47a0aad7c970b37397cd2402e:action", "state_id": "3527ce27d6e99394db6a12abc4b943e5dd64ad3af98318f2e99eabd7d9d92e5a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.24609375, 0.9609375, -1.29296875, 0.64453125], "student_probs": [0.21061742305755615, 0.43047407269477844, 0.04519474878907204, 0.3137137293815613], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2838858f278aa009e521ac8a2fc1e53a5d75aa7a924599d273420ff2c9f07ae1:action", "state_id": "4be8bb067481f50a73289f6762c86570ed058ecdb14ee6ce2d9144f0c2e4d9b8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0390625, 1.484375, -1.59375, 0.5625], "student_probs": [0.14032374322414398, 0.5954186320304871, 0.027416355907917023, 0.23684117197990417], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0c5882d128b74950ed396b551e442933dede1dd45b8621178d571e3ac1770341:action", "state_id": "c8d6e796c09953597a42a73a2100b5e874e769aeffa254e3b66ebc796fe2870b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.79931640625, 1.978515625, -1.54296875, 0.05859375], "student_probs": [0.050206564366817474, 0.8075280785560608, 0.023866921663284302, 0.11839848011732101], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bf568f2291f6f32b8d014602d5379ff6572820e4227b590e554559cfcc7e58c8:action", "state_id": "b9fe075b4694b98cd83ee8900dfaa5b3b070d2d9dc27a2a575c4be6060ef8a6d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.44140625, 3.046875, -2.0, -0.861328125], "student_probs": [0.010831114836037159, 0.9636269807815552, 0.006195537280291319, 0.019346298649907112], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3775dcde2fea3c63239f877f5a5b8ec24f1655d82867eb50b3f4df8788c611aa:action", "state_id": "e1fb7594cac2c9f58d75b9bdc06b23b001a020094fb11c2272dd72ff68351682", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0390625, 3.013671875, -2.171875, -1.44921875], "student_probs": [0.006244964431971312, 0.977022647857666, 0.005468274001032114, 0.011264083907008171], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "95e9d3d1b6738cd7f1bf318f58f3126094e484555e76e356a23b800bb7985828:action", "state_id": "06d3afd8ea097ce8ce493d89f928c0294fd53d036c7a390bab9f14b34d44bc1e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.08203125, 3.00390625, -2.0390625, -1.5546875], "student_probs": [0.006043397355824709, 0.9774076342582703, 0.00630873441696167, 0.010240086354315281], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e65a668b8b39ced439d87479c795c085ab7e47a466ab2fd12610b93b1a676b7b:action", "state_id": "0a8f9b43bb7972a518206659828d0958ca1e9e8b6fea16b32d0ccd6c7b359656", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.125, 3.056640625, -2.1875, -1.6015625], "student_probs": [0.005506554152816534, 0.9800264835357666, 0.005172928795218468, 0.009294068440794945], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f3ea3efeae61659c95b91c3c42ba2aebb1b8d796d2b64037d961476244912832:action", "state_id": "98feaa8f46c7b03d26700bcc023dbaa05e0a618f0661fcf6e87870604178664e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.18359375, 2.94140625, -2.15234375, -1.5703125], "student_probs": [0.005812184885144234, 0.9774591326713562, 0.005996683146804571, 0.010732084512710571], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0fdd38979939dad9c1ff188576a0aa6804f690b34ba01bc8cc35f47ccb5401e:action", "state_id": "4c4e4a14b03237bbaeb6b4d7d9e4cf13114d617cabe1adeda7b3519ba89a5948", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55078125, 2.8203125, -1.41015625, -0.6646728515625], "student_probs": [0.011946461163461208, 0.9453251361846924, 0.013750294223427773, 0.028978193178772926], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1200c9bf6d93e0ea60c8d63244a115e1c4cfb85bb0ef0413b4461ddc9b1d2ef2:action", "state_id": "18ff6c3f0b475352b979c787c4aee7b13e65b004b5b8d0518ab1f2699f44dfaa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.70703125, 2.87890625, -1.76171875, -1.19140625], "student_probs": [0.00983123853802681, 0.9643964767456055, 0.00930803082883358, 0.016464225947856903], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aee26c1fc701fa05bb986773d88073e2d3ed8872e250e3ab3e98de5b174e792b:action", "state_id": "559a1b8a1c1822a0a1b86d7af3cb348b29b937bf49bba464dd5fe0f902aaeb62", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.70703125, 2.732421875, -2.17578125, -1.087890625], "student_probs": [0.011336363852024078, 0.9605141282081604, 0.0070941150188446045, 0.02105538174510002], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "33d68def4ae6e361b967138fc3671897cfbe5186082b3765c88527e63bd25eaf:action", "state_id": "8313bf47872430601c756d28242d96447b3df70071dd993f60dc006ea42c716e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0234375, 2.71875, -2.57421875, -1.64453125], "student_probs": [0.008494589477777481, 0.9742003083229065, 0.004897124134004116, 0.012407928705215454], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "45038a852428957436dc806fb9567869c880a5d871a9ddccec9f933cf9465b59:action", "state_id": "57b5f12b0c92637be4a47bd4ece7675c9f9d917b51f14d9a910f8b147aa65c45", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6171875, 2.89453125, -2.234375, -1.35546875], "student_probs": [0.010647717863321304, 0.9697751998901367, 0.005744013004004955, 0.013833099976181984], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "69e02d9a0aeda9eea5ebaa3ecf3f6a094e777bad4db2a5db6f9ef2a1953e389a:action", "state_id": "73e2a3878a0758d90d631053be35a8887b06302e7a52319545af0d424411d29e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.40234375, 3.158203125, -1.7578125, -1.09375], "student_probs": [0.010131905786693096, 0.9689725041389465, 0.007100893650203943, 0.013794681057333946], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4dd8be92ce195bfc937ccef18707a4a5eeeb4cf392efa7eaae23e19465ded6c3:action", "state_id": "74350fb2d7fc3c863a81b72f81f3cede8d98e3c30a56ad376b745d3ae563dbf4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.96875, 3.15625, -2.0078125, -1.671875], "student_probs": [0.005831533577293158, 0.9807131290435791, 0.0056081307120621204, 0.007847186177968979], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5113754095b81b02d42198e54c62bdfc7fcdadefa28fb31a110f3f6d0f0dfd4b:action", "state_id": "3f456340d70c218b4e7e12b97aab639118e8010b907812cc10d3aecba6dfab1a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.88671875, 3.107421875, -2.0859375, -1.69140625], "student_probs": [0.006640937179327011, 0.979844331741333, 0.005441389046609402, 0.008073326200246811], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ab429d1c55900101e1d972108d95e4b763f2c809d87e3770769097eb000b8893:action", "state_id": "72640df1dff0ce961e69d4dbf5cf6a900a431915a697c80bcb5a67b973543676", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6328125, 3.154296875, -1.8359375, -1.35546875], "student_probs": [0.008124141953885555, 0.9745243191719055, 0.006630731280893087, 0.010720779187977314], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "27520aa042ae625da3e7c6d57761e1222328495102438e267f2889a57c9f6491:action", "state_id": "ddb7135f7a6e9f1a490226279f6280aadcd0728c85bbaa46d44165c5f2078275", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.87890625, 3.064453125, -1.69921875, -1.203125], "student_probs": [0.006925064604729414, 0.9711750745773315, 0.008288217708468437, 0.013611684553325176], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5a23478b2cfdaab338f27aaccf32e1d4d79b4ef5b2d0ac458a5dbfba9b8ceab7:action", "state_id": "f8ddb51b98865055d71fca34836e1b0dd3e5d82bab0036255a6223fd735c19bb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.76171875, 3.130859375, -1.4140625, -0.97265625], "student_probs": [0.007250902708619833, 0.9665220379829407, 0.010265433229506016, 0.01596164144575596], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "794262b15624a662131c3b6d5e46eb6f99e0ab44f5b6b41cd1c1bffe3419f28f:action", "state_id": "af6c549a838424f1ee0c0049f75843499c445eee0a617c2a6f5c25a683b5c9c6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.05078125, 3.19140625, -1.43359375, -1.1796875], "student_probs": [0.005145978182554245, 0.9730184674263, 0.009539136663079262, 0.012296433560550213], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5b855df64c47db872921768198b1f667af2442c23865abf93a79eae6ff66ee5a:action", "state_id": "de19992bf04517c25bdb611eeb88f3f60075560f554010b48b2e2c5108b00137", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.23828125, 3.251953125, -1.58203125, -1.3515625], "student_probs": [0.00403765169903636, 0.9783796072006226, 0.007782778237015009, 0.009799997322261333], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6c18889b7864ee76a7467b5edc0f96fe9dbb0ca2a43ff585c80ceccc613101f5:action", "state_id": "f24a0a448d8f078e48b9e68a105d1e94e08f3434fee0fcca415e4e660551ed29", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.31640625, 3.244140625, -1.984375, -1.54296875], "student_probs": [0.0037803472951054573, 0.9827578663825989, 0.0052690342999994755, 0.00819278135895729], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "de9886990bb09c17ccb63e3036be453153f5462983e246cd97cbdd874315e815:action", "state_id": "1c7b37b1ae4ad20c19e6bd4ff498c406cafe8cfaf64bfb35b83a88e7dbbf09e5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.18359375, 3.271484375, -1.73828125, -1.4296875], "student_probs": [0.004190598614513874, 0.980361819267273, 0.0065414318814873695, 0.008906220085918903], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "86ab791785a2b766676bd30afb9f9797e82dcc2dcc0b700bf5ffc020d46abe09:action", "state_id": "1a912a3d5daa8aac282a84dd4bcdd3bf4d8e6f5c9faaac94cdc523b3c5970c7c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.328125, 3.099609375, -1.8671875, -1.55078125], "student_probs": [0.00430303392931819, 0.9795122742652893, 0.006822717841714621, 0.009362048469483852], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "35404ce70cddadbcc37c7f3aef4fa32ba0fcda5a31008b9af647a76609094240:action", "state_id": "bdf1a60a5684588425a41dae44766de4c081090932690b6365890c65a17a0521", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.80078125, 2.98828125, -1.65625, -1.41796875], "student_probs": [0.008076857775449753, 0.9707465171813965, 0.009332791902124882, 0.011843928135931492], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "30829a2c2260c3491f784575c5d827f9ccf6b6f434355a0c53997514530bf8cf:action", "state_id": "e331312b74a82933451a48d957224aa23b9987d710f4d8039ca67e592b0b756e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.94921875, 2.990234375, -1.6796875, -1.4765625], "student_probs": [0.006963428109884262, 0.9727479815483093, 0.009117568843066692, 0.011171078309416771], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f1c37a5dbefda3e207769de68b69d5fe0f268c706ae3bdbe832098982d97be8f:action", "state_id": "f3ef0878ad14ff4cab1168a8f366aa6ff4d15080434dfd73ec06ec6ebcba68ee", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.984375, 2.98828125, -1.46875, -1.26171875], "student_probs": [0.006704904139041901, 0.9682549834251404, 0.011228601448237896, 0.013811415061354637], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e544e05b6f88fd214b640b4f7950e5dd4b6b858a32c95aa4e950c011f7e764ae:action", "state_id": "56ce18934c2e1bf2ad4120580148a4a549c22b050a3a03ebc6b8576e355bb041", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.73828125, 3.060546875, -1.08984375, -0.7939453125], "student_probs": [0.007883217185735703, 0.956771194934845, 0.015077048912644386, 0.020268583670258522], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bb518ea1ad9a9d181e34122f02ad76f1ae5461c94c4baa9bd0b42e668dd4b7be:action", "state_id": "fac1ce878bcdb024897bffc689bd7e2deddf66f50386bf8825610e51863aff52", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.46484375, 2.69921875, -1.52734375, -0.8671875], "student_probs": [0.014686527661979198, 0.9448187351226807, 0.013796715997159481, 0.026697952300310135], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "022a900c559388750c7da0df5adab7bec2f03af94d10a7bf5f286263befcc738:action", "state_id": "75d6553de4edf69407f6328d3c37163cbfa858f16af9bcb5e8b6f2cd111d6a11", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.7015380859375, 2.26953125, -1.5390625, 0.00390625], "student_probs": [0.043534476310014725, 0.8494784235954285, 0.01884087547659874, 0.08814626187086105], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "94f0cb6cfee218ed041e081b838d9b0db367bc17122e81f759a686c03a5bf2e8:action", "state_id": "3d63f74c3c375d9405c3f63dcc02f4d614eb146b3f0df873b4cfc7c4640b7e66", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.138671875, 2.28515625, -1.37109375, 0.5390625], "student_probs": [0.06872857362031937, 0.7758764028549194, 0.020040258765220642, 0.13535480201244354], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ac8b8f40a5af8fa28a2831db5d4e77bc3e73c9c66322d2559b4ab88d892ccd07:action", "state_id": "bf9da1522197044c16444a7175961773d60b52aa338c8a448a3e2df3da5b8753", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.01171875, 2.107421875, -0.9755859375, 0.90234375], "student_probs": [0.08374938368797302, 0.680979311466217, 0.03120330721139908, 0.2040681093931198], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9c6c90877885f595a04285e303a4dddf3f2ca3b7d7e718ad84736b72a0930b9c:action", "state_id": "5e16e47f9e24129456f6f07f98b203723ff1ad27ab1a7efd8ec892438d694a20", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.51953125, 2.27734375, -0.53515625, 1.1171875], "student_probs": [0.11153381317853928, 0.6468667387962341, 0.03884736821055412, 0.20275209844112396], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "281188e0a8b1e0367b3717e131f73896911db1adbd50fedede0f3149d57acfab:action", "state_id": "a7e05629bff780a403cff7d1926f099687e2deeaec3e2e3af74b2cccff942163", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.6328125, 2.166015625, -0.4423828125, 0.984375], "student_probs": [0.13521715998649597, 0.6264601945877075, 0.04614030197262764, 0.19218234717845917], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "440fc185b5f42f6ee9a328dbb488fbdfed7943b36bcf86cbd8c4c13823607214:action", "state_id": "b74b67f54efc71ceb05cbc8310e8f08358926fe7467e3c4543a7521a28f5e577", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6591796875, -3.6796875, -0.880859375, -0.634765625], "student_probs": [0.34786689281463623, 0.016967708244919777, 0.27870118618011475, 0.3564642369747162], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d2a7dfddfcd207afc75ffa8b69143575b10ef99d003b79e7b23bb9289cc4e9f8:action", "state_id": "faad4d0980eb7b85aecde02f09f2fc6610b9334325727b280876e9367c3c8419", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.958984375, -3.546875, -0.908203125, -1.21484375], "student_probs": [0.3446466624736786, 0.02591000497341156, 0.36260026693344116, 0.2668429911136627], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "79ae0d376c5034b4aa77fecfb7ee5c7ed904b7448b8593ebccb7acd752dcfea3:action", "state_id": "53095abb6eff771f0379d38000e74740f92013b4786031d08f6d0496a55a8ce6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.46484375, 2.201171875, -1.20703125, -1.3203125], "student_probs": [0.023504294455051422, 0.9189197421073914, 0.030416816473007202, 0.027159161865711212], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3c81917e8451195b38a2bf32fefcfc37abd1c8c49050d77567881fa12c5cc945:action", "state_id": "9d563105bb84cc77ad4766106789931cfff8e290c29e7ee164bd98c3ace3d272", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.880859375, 0.58203125, 0.2265625, 0.697265625], "student_probs": [0.32323041558265686, 0.23973575234413147, 0.16801756620407104, 0.26901620626449585], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "66253e4045a40b5318cbfc7ea15289c9bb277996ed0cdf90cfc7ee6fd02cc240:action", "state_id": "75e35ba8af88f317847b7e7bc8ccacaef89f8d4cfe9804752fbcf3deec788f65", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.892578125, 0.763671875, 0.61328125, 1.58203125], "student_probs": [0.4283268451690674, 0.13851523399353027, 0.11917460709810257, 0.3139832615852356], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3d4b9cd65fd4f03252daa2d9c89e780c081f59902394b7114d0547e2dc6b897a:action", "state_id": "fe8263297a8da17977a5ae254fdd7b0c4173a3d2919b479b0b8de4f76ec5f420", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.7841796875, 1.138671875, 0.7734375, 1.83984375], "student_probs": [0.3394908607006073, 0.17802771925926208, 0.12355726957321167, 0.35892415046691895], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d73e0e0cc7f8194591e3ab2aea0b5d9a1e9e3f50241a44ed498784271a86d9b2:action", "state_id": "6c8322f4ed0903a7e9a2952ab03c2e8727a63bf2103892222d6970f066cb7d14", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.2265625, -2.921875, -0.46337890625, -0.03125], "student_probs": [0.4315432608127594, 0.018521463498473167, 0.21646444499492645, 0.33347079157829285], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7ec3684d33dccbe74fa73a035c1d8ada3b059433d6ae9cf72001fc28cfdf9f96:action", "state_id": "bcea24b03b10afc894a59c2408909c70854d512df302d66cc255597bd7c8bcb8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.326171875, -3.15234375, -0.224609375, -0.46533203125], "student_probs": [0.3293561637401581, 0.019510794430971146, 0.36456403136253357, 0.2865690588951111], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6fbb9174938bada1c0227c6f7c0e94cd5d8e3cfe7ad5c54c8d986ec546d41a20:action", "state_id": "6daffb585e99e41bfec5845614310942a4ad45f40cde941357b7a0698f408df3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.89453125, 2.169921875, -1.87109375, -1.7890625], "student_probs": [0.016295140609145164, 0.9489157795906067, 0.016681568697094917, 0.018107671290636063], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "091269d07a4b32b5a5331d23c0a72ff334ecca408640f28685f947bd8f051186:action", "state_id": "bdce7b8b9d725c1117062186cc4cc62fbc2cc3b98bfd0187fd67193222aea9ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.03125, 0.896484375, -0.19921875, 0.71875], "student_probs": [0.15405581891536713, 0.38957226276397705, 0.13023574650287628, 0.32613617181777954], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e13d3f36e1ff95f2f87f0f602fd85aec5f1286e303b1b9901ed90edad4a67be8:action", "state_id": "ec24640c03ea8fc12143a876ef83c71754387421e00ab915f5cf855bd63d93ac", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.639404296875, -3.734375, -1.9140625, -0.71923828125], "student_probs": [0.44482553005218506, 0.02014007233083248, 0.12434052675962448, 0.4106939136981964], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4436727e29f6494e9bd1deb5ec3971aa100aa50432661e0a5aa9544a7a23a5cf:action", "state_id": "3d11d0dff6e5e68675949c2b331c2f693bf63944c6d3c460a9fc5074ec76ff05", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.2890625, -2.84765625, -0.636474609375, -0.52099609375], "student_probs": [0.38805919885635376, 0.030041031539440155, 0.27416929602622986, 0.30773046612739563], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fbb88dc29db24bb116461a8a082a86816ef9aa779ff8d7e1d5f9c0787f2737d4:action", "state_id": "57233f8351b89197a50e1c21a736040578f8ea57f3a25ea083e562d37ab9cf18", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.630859375, 0.015625, 1.7236328125, 0.376953125], "student_probs": [0.18872150778770447, 0.10200665146112442, 0.5628684163093567, 0.1464034616947174], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "52cefa4c9a06aa43a40e07db1576f3d7696f794da6712e955b9de6c3376fc1d7:action", "state_id": "894b1087d9cbe126c643104678efa32e29b22dce2baeff913fb1bfa816dcb1e5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.4140625, -0.1015625, 0.00390625, 1.1640625], "student_probs": [0.4459156095981598, 0.097954660654068, 0.10885029286146164, 0.34727942943573], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c0547490121179ad360405241dadb8aa2d023b23a28e991aef0dff63b5db5df4:action", "state_id": "5f41befdfc36e1faad0d1bd0daf003ff98484060e4d730f8d1e8946839a1b21e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.8671875, -0.251953125, 0.0390625, 1.158203125], "student_probs": [0.32245439291000366, 0.10530081391334534, 0.1408699005842209, 0.43137481808662415], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "24919e511d9e0a822ab404ece88155dfe2544d7b844b48c7cc0f1f97f68b9c4a:action", "state_id": "b71cb7a16f8bd00df20c999d3c8875e6b945d3050e1eb9eebc05b2de11f0dc90", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.37890625, -0.08203125, -0.353515625, 0.75], "student_probs": [0.15471170842647552, 0.20818735659122467, 0.15869022905826569, 0.4784107506275177], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7befd58dde3d8829cc31c85bdf45f3dced2e9ef44bdf148290a7bdc1eb8eeaae:action", "state_id": "c1af9a0ff2e036f8270f302e8ca3b5bbd35b07d43887b44b48c5f53412f37615", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.453125, -3.359375, -1.15234375, -0.53271484375], "student_probs": [0.19960595667362213, 0.02966877818107605, 0.2696504592895508, 0.5010747909545898], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5750b9712f17b7c5513f1da2486093033e302afd5286252754df8c3a1ac69672:action", "state_id": "cf666e730701cf3e898d85afe4e317ac023426710dceebae5d62bbc52976c36f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.015625, -2.9609375, -1.017578125, -1.0], "student_probs": [0.1457168459892273, 0.056619465351104736, 0.3953265845775604, 0.40233710408210754], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9fd19e97ec6ad10e59644501279d96834a052c74d6fc0371edf13f570b21d6ac:action", "state_id": "07481b9c47b85919976d3c71849b96692e8b0339b537d80f5dd6281cbe4a9e13", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.43359375, 2.578125, -1.62890625, -1.86328125], "student_probs": [0.0064446525648236275, 0.9677457809448242, 0.014410226605832577, 0.011399428360164165], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cab545c6a2049cddb62d966ae25b751ac675b0e03284a023e78e1d22d4d03b81:action", "state_id": "8342888de1c4e9f81ace300eee5738ef1839766e712faa5cd3a1410cdb5e88a6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.890625, 1.0078125, -1.8828125, -1.18359375], "student_probs": [0.0450824610888958, 0.8180559873580933, 0.04543604701757431, 0.09142550826072693], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d43fca306268cdc990668e681d5b54e8209bfacb2626a4a98047f27ec3ca05ed:action", "state_id": "1706f0363f2dc050a153eecbd0ce529b03e61ab2b73ec7b7979d4f36f44b6a33", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.23828125, 1.591796875, -1.71875, -0.353515625], "student_probs": [0.04764696955680847, 0.8074629306793213, 0.029469337314367294, 0.11542081087827682], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c2338e37e94de2dc86ad87f825813fd0fabd0883e092f1a515f15100eca3f61d:action", "state_id": "063d83c36cd3a76d07a425f94721e1eeff40d6737b6bbfa3467757c87a6786e1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.46484375, 1.845703125, -0.7127685546875, 0.748046875], "student_probs": [0.15120114386081696, 0.6015263795852661, 0.04657196253538132, 0.200700581073761], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e6e00c6a0373c11106937b2f250049d898508d9b40c0d9c19776ddc88858b99a:action", "state_id": "718d8955f216e54de3c67ed262008bd4685660a73a81eb9e2e64a1f05dcb2531", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.017578125, 2.9296875, -1.68359375, -0.48046875], "student_probs": [0.018175754696130753, 0.941386878490448, 0.009337821044027805, 0.031099693849682808], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fac70a936b4cbadad0b820efbfcfb2053f09a1fa74e5dcacca2181d66aeaf37e:action", "state_id": "0543f9c87fdf2b6b7ce3892b10955fd0845cc022dbb82b4aa1e0334a4a8e9538", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8349609375, -3.6171875, -2.35546875, -0.8134765625], "student_probs": [0.4343636929988861, 0.02688734233379364, 0.09495227783918381, 0.44379669427871704], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "70d5a6493420536edb042e8eb1ae77b1d724eef981060eec4574668c7f8e4e82:action", "state_id": "b3eee8c93fdf19d60d4cf117eafbcba403cca2bdfd81f37febee7b9d880d99ef", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.291015625, -2.67578125, -0.83984375, -0.5087890625], "student_probs": [0.40419644117355347, 0.03723076358437538, 0.2334745079278946, 0.32509827613830566], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "af98c5e9c7045e9663acbb0b73a1f875949413894de23663fa89e821f940f6ac:action", "state_id": "e32636dbdeea1d586024cb5878c7772cabaa1e0b03ff218d3ee01988e29c2bb5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0, 2.552734375, -0.4794921875, -0.900390625], "student_probs": [0.00966472364962101, 0.9171003699302673, 0.0442117415368557, 0.029023095965385437], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "91223634a648920103ec5d929d041cc6454bca5d6ef22e15c510bc2f6c4cb91e:action", "state_id": "e16f5531c27baa6df2adc9b90359d00ea2aa44d9076b770d6d030ee04b065803", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.078125, 0.4296875, -0.15234375, 0.30078125], "student_probs": [0.08325538039207458, 0.3760511875152588, 0.210123211145401, 0.33057019114494324], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "31cf12f60d07e32a2708f299621d3542062105bd4c6b4a4a7df6b6f0a721be96:action", "state_id": "22eca9dfb8c7c162c0d20e8c030f665879a098c03efcceeb78d290acd93abfd9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.66796875, 1.265625, -0.533203125, -0.046875], "student_probs": [0.03576010838150978, 0.6721131801605225, 0.11122983694076538, 0.1808968186378479], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3e0a9104c05803c67e0171e98ad3985279cf4addec4fbeb625b7d683219946a6:action", "state_id": "1e9aecd379acc1e914c4394ccc0b76956307b8bc81d20a28325f542977b67711", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6953125, 1.626953125, -0.705078125, -0.224609375], "student_probs": [0.02795853652060032, 0.7750969529151917, 0.07526060938835144, 0.121683768928051], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "59cdc549d0b13a82accbd441d0122767873646a44410ccb7d27795cf004e0127:action", "state_id": "f3262f3db92183ef98116f75160772d734a99cc39571ced61ef3596260c9b176", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.03515625, 3.029296875, -1.296875, -1.048828125], "student_probs": [0.006095050368458033, 0.964809000492096, 0.012752894312143326, 0.016343088820576668], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7ac54d0e9e0c67fb564c6f9d363413dbe1a7df96ccf6ce96bb4e870dddfda62c:action", "state_id": "5b956f3e2c8bedce5acac1d26e7c070b38008ef8605eb2a6520cacd1f4cd369e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2421875, 3.05859375, -1.45703125, -1.3515625], "student_probs": [0.004851477220654488, 0.9726890921592712, 0.010638074949383736, 0.011821362189948559], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9c859f80463b88343717182bf5633a51b4c73363d7cfaf4d9b1494f0e772b34e:action", "state_id": "cfd58ce5df5eb34f3380b1a4c0463a62f529fc41a7ffbfaceb9e42641d767fcf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3203125, 3.068359375, -1.6328125, -1.55078125], "student_probs": [0.00446309195831418, 0.9770264029502869, 0.008875918574631214, 0.009634718298912048], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "90702a668f7ba62e8afef41452765eb003235df1257211d930d6b66f4ba50636:action", "state_id": "e794b6079f592a9bc007e365afaeab7d70c60bf51f92f7df8f54702544ccc899", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.4296875, 3.07421875, -1.66015625, -1.6171875], "student_probs": [0.003983080852776766, 0.978442370891571, 0.008598492480814457, 0.008976012468338013], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d4e5bd8ef81db48021667a906c24216d0e205897a8e5871e47dfd03f314b7f5c:action", "state_id": "ff7293348aebed878f63a1e4960282b84985c06e3b51a287abba94a23d851cf5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.36328125, 3.095703125, -1.625, -1.51171875], "student_probs": [0.004161561839282513, 0.977379322052002, 0.008707386441528797, 0.009751810692250729], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "48432c7bc95bbd61185978e229622e800159ce5d440598bdcc456089f8b7a615:action", "state_id": "729d1e4de206634dfc63365e0ba2f254631ff0d0677595251282e8b6773aeba5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.33984375, 3.14453125, -1.57421875, -1.47265625], "student_probs": [0.00405796617269516, 0.9775573015213013, 0.008725997991859913, 0.009658800438046455], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4a53365a4851041067da19dd80876ca1350a1123f9f6b32f7530925b0f9cf42a:action", "state_id": "e8153ed63c2c683bf24c0b1ec9ff3cad3c9ae1d6664cf9d50c51380cb4faac7e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.359375, 3.09375, -1.6796875, -1.46875], "student_probs": [0.004185911733657122, 0.9773545861244202, 0.00825989618897438, 0.010199611075222492], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "34a0f49a44a84f0a536cb6e181e82b83342c4c75f5089f7d902140259c6c050b:action", "state_id": "a2681c747fde7620e937f228262532a1f43d99367d08f4f205b0027c582a2e1c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.359375, 3.07421875, -1.6484375, -1.51953125], "student_probs": [0.004267621785402298, 0.9771600365638733, 0.008688447065651417, 0.00988383311778307], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "11d39ab21fceb22a992f669415c13551b005809ec55d39164d56d7c236615830:action", "state_id": "092adc3384b10ecec8cffd0fb1d592c48f642f074708e8f877b16ad4036955c5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.34765625, 3.11328125, -1.6328125, -1.359375], "student_probs": [0.004148544277995825, 0.9762268662452698, 0.008479074575006962, 0.011145533062517643], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3d55e2fd31aff441590747c68f7f0a8fc55ed07b836d1efa793b744f8a17ec20:action", "state_id": "60a4c5a3a0f8cdb352c4fb7dbc0d69adebfb38de57b7666d468b51a1b6ea17ca", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3828125, 3.126953125, -1.58203125, -1.35546875], "student_probs": [0.003950787242501974, 0.9762126803398132, 0.008799510076642036, 0.011037060059607029], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "de15eb8cf167f95f14f0f4dcce4d0eba376aec9b831069f270aecb563487f5b9:action", "state_id": "c5a7cc38dea234977c96ced2b8242f6e58578a6fafefb60a6da132d02daae17f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.38671875, 3.17578125, -1.48046875, -1.33203125], "student_probs": [0.0037478546146303415, 0.9762157201766968, 0.009276029653847218, 0.010760381817817688], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ee0ac6b02ebce55d0ec5432dd70e0dd6a19e38fecd905cee098fd319be64dc9b:action", "state_id": "01620d22c98c4f86443cea6f7a3d1a639ca52598506b90c6a69fb7821a5a0149", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.39453125, 3.126953125, -1.53515625, -1.44140625], "student_probs": [0.003906839527189732, 0.9767327904701233, 0.009226720780134201, 0.010133570060133934], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6ef8f1dcb2ec16d439527372ca134df537bf1fa8a359a69efdc1d2401413bd56:action", "state_id": "b5a78962b4633a9d16827cf3c42025e61c05f3b95f5fdad87b64930aff0decf5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.30078125, 3.150390625, -1.42578125, -1.3515625], "student_probs": [0.004183861427009106, 0.9749698042869568, 0.010036562569439411, 0.010809803381562233], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1abd103fd4d470a97845f3c299601dda0b8253bc68e7d21bc06c89169321b545:action", "state_id": "300839f97fcb53a3df19cc87db50b8bf92dd6acfc8c38772f6b9e8f121da214d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.26953125, 3.15234375, -1.375, -1.31640625], "student_probs": [0.0043039810843765736, 0.9740040302276611, 0.010528351180255413, 0.011163678020238876], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81d821d11d5adc4d8437c8b9aea229b28cc0c9ba63d6feb7a5fc0394152ddc32:action", "state_id": "ea92ce82d54837c4ae5eb89553ad7b63f9fba3238b6b536e7add8bfc94a673c6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.14453125, 3.177734375, -1.23828125, -1.0625], "student_probs": [0.004733209032565355, 0.9695858955383301, 0.011714804917573929, 0.013966123573482037], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0b6eedeaa3858bc3489c2fa809a8cddebb0958ab8410af7ff4d4560a4cae3200:action", "state_id": "9ba3b260a24e3f88f10b45d9b3df80193708b4e51a42eecb7200a8e4b268fe9c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0703125, 3.234375, -0.99609375, -0.912109375], "student_probs": [0.004798694513738155, 0.9658721089363098, 0.014049161225557327, 0.015280034393072128], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cd0967ea743b65731572f6e767c1a7c5969726f5e04f18b8ee206da612dc860b:action", "state_id": "c075e1cf398822731af5227f8c5905cdf400d498f17e953774bf53a3c095f9a2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.02734375, 3.205078125, -1.0234375, -0.92578125], "student_probs": [0.005155076738446951, 0.9652661681175232, 0.014067796990275383, 0.0155109241604805], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f986af71df8837c73b9f974cb6ca4936d6a8d0420b531c642456846cd0dd0c74:action", "state_id": "88ba12e91233d0d95974678ad43ecde6359c2359b1d8eefafe550daccb7531cf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9375, 3.2265625, -0.869140625, -0.7421875], "student_probs": [0.005491830874234438, 0.9603753685951233, 0.01598452590405941, 0.018148250877857208], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "77d7badd68b9ee9df3d8f78f774c59021dc6f56312abedbf57453d24bcb6973d:action", "state_id": "6117b1f44383ec2c6ea17709f85a9ef748a78dca418d3bd910d21a20a5e5e0ae", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8125, 3.26953125, -0.72021484375, -0.5574951171875], "student_probs": [0.005931553430855274, 0.9555789232254028, 0.017682425677776337, 0.02080703154206276], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ea76833a671612af4a946c5a381aa80e439dc11be9d5a032dbcf1268720d75ea:action", "state_id": "76286baeb622e930932771d9bc7fc47b1fdebb7981fd55161240aa3a0c20ec5d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78125, 3.294921875, -0.529052734375, -0.171875], "student_probs": [0.005894229747354984, 0.9440184235572815, 0.02061813697218895, 0.029469292610883713], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "36362946a66a0c09828175a20d66ecd5448199739a4b2d997ecac3f7c388edf3:action", "state_id": "0b225518abbb6c9cf56af191a028d274bff415583c75de5e8152e44b22230eb0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6484375, 3.15234375, -0.47998046875, -0.05078125], "student_probs": [0.007647373713552952, 0.9299618005752563, 0.02460179291665554, 0.03778901696205139], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "07c8d45f0ec57b3068e00b0606e692ed80d2b6a8274a672471dfcab5cf506f2c:action", "state_id": "520bd2cf858029e2dfa93cb769722053864314c4b3e57613a2a6da381dd034c0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.44921875, 3.177734375, -0.220703125, 0.06640625], "student_probs": [0.008995183743536472, 0.919327437877655, 0.030728939920663834, 0.04094846174120903], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0f72b5131cb2f2073d790d1a089ffe4b3c9fe4fbea3337cab89d47642bd94318:action", "state_id": "948272d33ebe24e82caf3163937d83966c2945b28ee123a735ea4ed46c911f2e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.41015625, 3.150390625, -0.275390625, 0.00390625], "student_probs": [0.00962845329195261, 0.9208244681358337, 0.029948769137263298, 0.0395982563495636], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a87f07f36802cb86b723b3d797c7f70541ccedfa2eaf4d7fc4e1b599c4acf0d2:action", "state_id": "0e596d12d407600c222c18d28beabdca6100e7e8e65cd0c27b48c9752be5244b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.95703125, -4.71875, -1.1875, -1.29296875], "student_probs": [0.08116507530212402, 0.0139400539919734, 0.476284921169281, 0.42861002683639526], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a469b9a2ee4483ca5f85c6894a6068715fca29a5f55e65b676675093bf3ed262:action", "state_id": "4d89d4eec0c9a94465b1d5c2800f482699053757f3756638c31dabf733422497", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.24609375, -3.765625, -1.1875, -1.001953125], "student_probs": [0.132081538438797, 0.028901346027851105, 0.38070058822631836, 0.45831653475761414], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "96cab600aaa79da16cce32a4e54f595db270151fdb2002d496a77e3a08a33ec0:action", "state_id": "4e52ffbbae2d5ce7fa595f6659288e25a51be00ad3d7e6523bb9a39d1c73af2d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.79296875, 2.60546875, -0.16015625, -0.6251220703125], "student_probs": [0.011030586436390877, 0.8970481157302856, 0.05645729601383209, 0.03546401858329773], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f34c48570bff50b48cfa757ce0e941d58a23e00be821cbe180bd19c3980c8986:action", "state_id": "7891d3721e2ea26c7a7f5f5907da2d2e85091515f06edd94cc9f60e870cf005d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9326171875, 0.37890625, -0.140625, 0.28515625], "student_probs": [0.09709426760673523, 0.3603968620300293, 0.21436379849910736, 0.3281450867652893], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "54eeb22ac571bbccd94df9a0ecd6fab41392307be562b0855ddb67a8c0d58a1c:action", "state_id": "fefbf7c241ac6f851f4b1cc9edfe848cdadb54ab8822360c44bae0c11ac502b5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.80859375, -4.1875, -1.20703125, -1.2265625], "student_probs": [0.0902734100818634, 0.022735707461833954, 0.4478262960910797, 0.43916454911231995], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1c64b3bc0d8cfa71adbe0685012185a67caabf95eaa0007c314150a519296633:action", "state_id": "4ca7db420bb97e1ebd4f5b7238cf289769037b7e37f1d1d406e88695cc23846e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.76171875, -3.6953125, -1.03125, -0.8349609375], "student_probs": [0.17400425672531128, 0.025165801867842674, 0.3612421452999115, 0.4395878314971924], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e9f12e79fd49ba14718fb25b362171d67b1d281a669fdaec12c77a42bfbf2baf:action", "state_id": "7f6e7ef33ccd29452b3502b463511e3746165ed67fac96bd0ef50d8e300df7ef", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.9609375, -2.9609375, -1.5390625, -2.01171875], "student_probs": [0.11456622928380966, 0.11456622928380966, 0.47486382722854614, 0.29600366950035095], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "71c286681014fd03e3e43448f144a25a5980491e40f37797b7868a54aff82588:action", "state_id": "6e3f5c2f7284fe6b07c5b3e4abddf3b9d17611b9d6599cc3b4a865b3031ff773", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.046875, -0.642578125, -0.23828125, 0.02734375], "student_probs": [0.13036616146564484, 0.19532091915607452, 0.2926393151283264, 0.3816736042499542], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "00ceb16256e0bb06dd77d1206b32f3d0f1c04b23fc905a871b1b0cdd0424b78b:action", "state_id": "b72ad0b9f601f7c6d875f6c37e9f683d2a5adccf374a15df1e5378cd1435a6f6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.15625, -3.875, -1.087890625, -1.013671875], "student_probs": [0.1384134590625763, 0.02481616660952568, 0.40286627411842346, 0.4339040517807007], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1452167ce5be2ef326f38cedb2350b1b059c9313ff7b1379717fa4bf60728566:action", "state_id": "108abd09f3ecc54fd267e4347b475b80ecbebc7b0cb026b6483c5dfee0d97237", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.12109375, -3.48828125, -0.7470703125, -0.551025390625], "student_probs": [0.2317119538784027, 0.02172160893678665, 0.33680981397628784, 0.4097566306591034], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1a8007265927fb808190e3da21ec5f37dd5960bff9989cf7610cfc6ffa9b884b:action", "state_id": "80a4a9430a59aa962b8dfe52bdca79c46158cb353f06d902083940b884dff593", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.25390625, 2.4609375, -1.84765625, -2.34375], "student_probs": [0.008695151656866074, 0.9703038930892944, 0.013052968308329582, 0.007948012091219425], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ce66ed5a8458e46f2bfca98f9b9957e3f09b5421f53fe4d52eb017af13a488ea:action", "state_id": "d062cb723d6168ae87e878addccc545ff42c7eb7605975e905f1aca6c9df5147", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.03125, 0.65234375, 1.01953125, 0.2890625], "student_probs": [0.13853435218334198, 0.27443432807922363, 0.3961922228336334, 0.19083911180496216], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "51a515a63122406b872a5922682e7c8c0a5afadd8990cc0003a0778a78c90932:action", "state_id": "c6360dc3526bdbe0a108e4db16dfd798f5ff65ac0d0d1c2a961822d4ad8ee2bc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2734375, 1.298828125, -1.04296875, -0.758056640625], "student_probs": [0.058723609894514084, 0.7690126895904541, 0.07394418865442276, 0.09831953048706055], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c1e629ce62d054a626887c008f0baecb328dd8392df862b18279cd4add0b0151:action", "state_id": "9550492839908ed8dca388d0295b2275592129b143d0438b79db41a861e4bd38", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.017578125, 1.87109375, -0.2890625, -0.39453125], "student_probs": [0.04365662485361099, 0.7844845652580261, 0.09045664221048355, 0.08140216767787933], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "647d56c18cd3201ea0f91f71994e85acf7e7e0bb64cf6f664d6f068b0df24988:action", "state_id": "edc74b325187ca4a7740c4b849f9d415e58fdc769c0cad883a3692a5f10082e9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.78515625, 2.8203125, -1.4609375, -1.13671875], "student_probs": [0.009585397318005562, 0.958825945854187, 0.013256123289465904, 0.018332552164793015], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6a96a95f96d4a5bc7d1d6422004b9904050a631c6583b5c521562e3a4609f9a2:action", "state_id": "01483ab660fd600534d7781a969dc043d43d24a631543db95c27ebbca0a9307a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3046875, 2.859375, -1.87109375, -1.66796875], "student_probs": [0.0055770426988601685, 0.9752766489982605, 0.008604217320680618, 0.010542107746005058], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "88ed46bc180f9a4f72273395d7bf781962769f9b115134b24f2c58956a8795b6:action", "state_id": "a20ad437ada0514b5186b8f8c94d0fd8456b6b49c5ea28c8483f3cbd5c3f0150", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.34375, 3.015625, -2.0859375, -1.76953125], "student_probs": [0.00461548613384366, 0.9812156558036804, 0.005972883198410273, 0.008195917122066021], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6d19e1460277c02a1988160b991553fe52ffea0fed9bfe9b2f632ee23ef4b4b7:action", "state_id": "610493077e803d591a5f80fce199fb5c1c26b03ee2bddf4e5f2db5f2ec3711c8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.45703125, 3.1015625, -2.14453125, -1.859375], "student_probs": [0.0037930163089185953, 0.9841273427009583, 0.005184438545256853, 0.006895146798342466], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a15302f9eef6aaa30ac6d218fe8dcd14ce77acccceaf3fe7303bf9f0415c2cdf:action", "state_id": "8d61c44873bbad68055f9ce72f9850402dc979ba32a892930a1723fc542fb156", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.4453125, 3.12109375, -1.9609375, -1.83984375], "student_probs": [0.0037601341027766466, 0.9832474589347839, 0.0061033000238239765, 0.0068889823742210865], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "776de58982d2631c704afa95f287ea22168eb8cf184a6b8fbbb33a8403d58c6b:action", "state_id": "0421314c55feaf3eab1c9cb5ae672020dd457946edca8f8caa9a68f30b0ba9d5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.5703125, 3.17578125, -2.18359375, -2.01171875], "student_probs": [0.0031527229584753513, 0.9866943359375, 0.004641257226467133, 0.005511629860848188], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c42886f298ba17593c31f2c58f029ca1b159c3fc4389ce0339a0e593ab5ff05c:action", "state_id": "5dc8977b8c4c863ec0b491c615812b01bdd6f7ce663cc6884239d85f39d03dbc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.06640625, 3.029296875, -2.15625, -1.85546875], "student_probs": [0.006007177289575338, 0.9810840487480164, 0.005491004791110754, 0.007417874410748482], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d9abf9d5e87d064049d8a8aa68bd23fa7474bac84f7842540819c87d1d57fa68:action", "state_id": "2458da58bb925a1d8a00a0e84c948ce4bf0b0efc4b717b8919b5215299e784ef", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.34765625, 3.09375, -2.375, -2.0546875], "student_probs": [0.0042720455676317215, 0.9858448505401611, 0.00415681442245841, 0.005726253613829613], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7a7486faa50b4d5418636fe06ae8a7ae0e8c5ae2b0c077d06d6b6cc27550f30e:action", "state_id": "5f307ab9e9d2ff53925c7646a549b3f91bba9cee8e5b97dbca6217ad74151d8b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.34375, 3.12890625, -2.1328125, -1.91796875], "student_probs": [0.004134667105972767, 0.9844302535057068, 0.0051056318916380405, 0.006329289637506008], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ce13731a21837ee2b7e802b3146aa309065452c657edbe64b4b95e0251449a9f:action", "state_id": "2ced0cf404831ec475ee31045e4b912176be1dadc30c9cf751cceb77a1d05fc2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.20703125, 3.205078125, -1.921875, -1.76953125], "student_probs": [0.004386299755424261, 0.9829864501953125, 0.005833646282553673, 0.006793634034693241], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8633a632cc0d4c3ec26f0e6d4e2e53a91ad95cf4bcf543ded2eae7b8ea14aef6:action", "state_id": "88ea277262df4c002ecbda90b8d99ed9c0b7a64b248395861c1c0920ddfe19b3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0625, 3.283203125, -1.546875, -1.44921875], "student_probs": [0.004667957313358784, 0.9788953065872192, 0.007817357778549194, 0.00861929077655077], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "905b061b2616c97caf020a9202dcc80eda8b8e285b54e1fd4a44e4a5ec052e11:action", "state_id": "38301837db265056b6e27f48a6aa82acfca7704791beba8d40cf8cf69e27ce0b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.65625, 3.21875, -1.16796875, -1.1640625], "student_probs": [0.007394286338239908, 0.9684603810310364, 0.012049086391925812, 0.012096244841814041], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f588227769828eea43f36e8cb7724568a8eeb44f330ebd509c4b2e107187ad6c:action", "state_id": "8e235ec28c09b9f81dc09c945bd0f455f063895ebef7a8cf7f970389748c763b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7734375, 3.0234375, -1.63671875, -1.23046875], "student_probs": [0.008000070229172707, 0.9690588116645813, 0.009172124788165092, 0.013768991455435753], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "843c2f9269cdaef3f450a598ae1c192742fb64aae3717444c939d6150bb4aa99:action", "state_id": "bfe1a64e6ae918ddb9d598cf91ac927e5da3f333093190c620083b68e4da8a34", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.06640625, 3.072265625, -1.96484375, -1.53125], "student_probs": [0.005737109575420618, 0.9781151413917542, 0.006350401323288679, 0.009797348640859127], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9ce73fd393d285ac48a168930c8662614ea7a4b677fbdf7f0ea21d50bd9d30d4:action", "state_id": "cead1cb498aaf5a7721a7d0bdfe8b06d15b49ed7cb168fcf5b1dba560177c134", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.896484375, 2.873046875, -2.109375, -0.314453125], "student_probs": [0.021530037745833397, 0.93353670835495, 0.006401666905730963, 0.03853166103363037], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6473682c3c2eac98773efb76aff00aa21739e89e8d3f21029985a1fb842ff487:action", "state_id": "3387fe218707ee1df1442f91b31105e412b0a07d788776fcf841729fa7284c42", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.755859375, 3.060546875, -2.0, -0.16015625], "student_probs": [0.020600248128175735, 0.9360878467559814, 0.00593675347045064, 0.037375155836343765], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7cb1dec85e6afbf3df73ca8f42f6d941def6da3f4920158469c283335408fa22:action", "state_id": "e45e93905bd87e89ab48d429f362515752f9c774fbfc407a1a88846cfee50211", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.00390625, 2.986328125, -1.37109375, 0.54296875], "student_probs": [0.04404744133353233, 0.8693007826805115, 0.011136936955153942, 0.07551489025354385], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4b021528c1ff2a8c8f646f63e3e3ffc6f7b09884a7f8b87ed53d5e8070b14f0f:action", "state_id": "44941e135876cd5eadd7c6a26680d94d82d3da00ae204525289fb31dab093b01", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.33203125, 2.990234375, -1.0078125, 0.55859375], "student_probs": [0.05957064405083656, 0.8501102328300476, 0.015600753016769886, 0.07471832633018494], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "97b4a6c77e327c99ca69749a140eec2978d379fff509c77f65c0e7d040fe475c:action", "state_id": "22219b2eb54be78d6774f24be873fd86dd17f7030554b81b28c98909bdb747ba", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.03125, 3.193359375, -1.16796875, 0.57421875], "student_probs": [0.03533976897597313, 0.8885743021965027, 0.011339476332068443, 0.06474637240171432], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ef8964c19c61e868c546c71a34eb7cfee9abf82e3e60d8d9fee02c1bd4a5fabe:action", "state_id": "d0279e7e85698126b16aef87cedde6807c8cd3938a039d001ab7a186616c3def", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.5078125, 2.87109375, -0.7074642181396484, 0.8828125], "student_probs": [0.0747530534863472, 0.7943080067634583, 0.022173842415213585, 0.10876505076885223], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9f5ffe41182713936aca183d9316f2f098bce0e0664a1fc797385751ffcf66fc:action", "state_id": "c21be05ae8d65f5859f8326a991f3719c6cb9a40b423bcc264e9b1385198b1bf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.14453125, 2.880859375, -1.064453125, 0.7421875], "student_probs": [0.05391830578446388, 0.8319714665412903, 0.0160946287214756, 0.09801556169986725], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "556d6db969288f2472ea2c79eaa2aa0d31136cf02e884f85608dc0ee484ba056:action", "state_id": "924ff51f81da73888e2243e41728a66f003c49308227b878bde2718c7c25ca66", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.373046875, -1.54296875, -0.749359130859375, -0.5732421875], "student_probs": [0.3551956117153168, 0.1102495864033699, 0.2438019961118698, 0.2907527685165405], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "917bf77081fb2e7db3c092aaa160d062038365b619241bab1f7a17034b636d09:action", "state_id": "cfa37b8b9a3bfff7c96b473e913c0fb0cb62fc5062575e27740b7a45c986f010", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.14453125, -1.9140625, -0.109375, -0.115234375], "student_probs": [0.373883455991745, 0.04771999269723892, 0.29004552960395813, 0.28835099935531616], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7bdc3f247c78a7fc01f8b8a3f6e010ac76268236e5089be8e6749435ad7322df:action", "state_id": "20f186c7915470019aa71e762678eb1100a9bcd6e603a83d01dd4fa89c6d01b5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.69921875, 2.59765625, -0.072265625, -0.55126953125], "student_probs": [0.012090449221432209, 0.888283371925354, 0.06152040883898735, 0.03810574486851692], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "93b43ad6cf0b4a143c6f6afaddd4ed9d519a6d1b3f5de124a02886a2c8660d9f:action", "state_id": "96968a62ee06cfdd4226f0557cf3c1ee0529a36876a60171d12476d19926ad3a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.86376953125, 0.4765625, -0.0390625, 0.34765625], "student_probs": [0.09560418128967285, 0.3652377724647522, 0.21809342503547668, 0.3210645914077759], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "452fe5e8370a7f586b84b44ec786be2178069edc5389b6bb8ba35bff4bbeeb0e:action", "state_id": "47d1d55df3ebdd1016ac8983e2e4b5579b7541a3d6efe95a89938e9aac21678e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.27734375, -4.3125, -1.115234375, -1.15234375], "student_probs": [0.13499747216701508, 0.017638780176639557, 0.4315422475337982, 0.41582146286964417], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "50ec9511efc421ecd05dca45d850830c28cb84650d9a65c5ec13deb46a953b8b:action", "state_id": "9cb87a714ae1322083794b63b9d90445454f647c7ac9a69555104b5f80aef2e0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.30078125, -3.703125, -1.01953125, -0.8291015625], "student_probs": [0.24888162314891815, 0.022525174543261528, 0.3297145664691925, 0.3988785743713379], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6355ae4b19040971240494266727f024cbc8d99d47fb310268f449d79a5d790c:action", "state_id": "6d820af95dd0a8cdc471c14598bf64d74f7c4475e399c863cb159581225cdd5c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.83984375, -2.796875, -0.685791015625, -1.90625], "student_probs": [0.07571592926979065, 0.07904025167226791, 0.6526501774787903, 0.19259360432624817], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8570455aa10de5cb1cd742412eb3cde493df0df40d5d298c64891a45649dda34:action", "state_id": "ec808598a8e8fe9a24108315a92efa479fd0c2674722d9e803f79fb6b801d710", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.36328125, 0.296875, -0.130859375, 0.171875], "student_probs": [0.06977535039186478, 0.3670275807380676, 0.2392963171005249, 0.3239006996154785], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d1ba381397f286f4e1f90400d6e70b909eb08554bdb80df824f437c092c4e7bb:action", "state_id": "aa1a633a268b373c3cda9e0fb3297a30de60901cedc2540620b31af5797d12ae", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.46875, 0.5546875, -0.140625, 0.1953125], "student_probs": [0.05675702914595604, 0.42932620644569397, 0.2141987830400467, 0.2997179925441742], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5e72b6313ec8ebe9451943cbeea9b1002ae9ef0306e47b84b7df15f41fb1a2ae:action", "state_id": "5f34b86849f346ed24e8bec1dc8adde1988dd8fc1365db1f7313d84d84e9844d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37890625, 1.169921875, 0.01953125, 0.31640625], "student_probs": [0.04293813183903694, 0.5492690801620483, 0.1738508641719818, 0.233941912651062], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d2ed2266ecb337e96b4c9308aa2008545f4cba9c3423fb7f953b1983e22f0359:action", "state_id": "6378d67c2bf434574a2a575e8443e486d47287425b536ed5d021565f25d8eab3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.515625, 3.1328125, -0.6442489624023438, -0.21875], "student_probs": [0.008971043862402439, 0.9367716908454895, 0.02144256979227066, 0.032814718782901764], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5a19e23aef5d752653600cbcb9a3120c225b3cf7dfb2284b6345254c94889b14:action", "state_id": "8c2bf8635178dada6d6017f3946ec4412eb447e245b3f2aa84b0e3ef5c73b734", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5234375, 3.1171875, -0.717529296875, -0.4443359375], "student_probs": [0.00910831056535244, 0.9437037110328674, 0.020391037687659264, 0.026796970516443253], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f4f8a1869020a9dff94d017a229d628b39df1d3ccd889cec150272c094e7480a:action", "state_id": "0387ccbcf4a4d4383d313e517bd4f131695001576616b3b2efa139098fa2510b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.90625, 3.052734375, -1.12890625, -0.921875], "student_probs": [0.006743048783391714, 0.9605408906936646, 0.014670753851532936, 0.01804533414542675], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9752be725d6d583ebe3ef571e3c2b8273359048453a8f83cb43f4156be742fa8:action", "state_id": "159d135a117876c9966f3c7140b425d33c5a5e7e1fc77d174146e4d3b4916bdb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.17578125, 3.05859375, -1.31640625, -1.234375], "student_probs": [0.005166968330740929, 0.9693843126296997, 0.012202748097479343, 0.013245957903563976], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "97d5787b26a66f939804a26fc5b1766dab66e58f12a47270f30a62455dbed609:action", "state_id": "7283918e6ecd621de1a9bd7ccf99c8f6212704c239421a56afe5cae953dfc131", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.13671875, 3.046875, -1.28515625, -1.20703125], "student_probs": [0.005428895819932222, 0.9680942296981812, 0.012721559964120388, 0.013755286112427711], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6fc2bbb1e224256a982dd9f1a849b98d5f48aa32b31a3641e7ba917517c0c2cd:action", "state_id": "23d07661e93313d3449adfb7520ec3ea44d1cac547cf987b80e29f425d29a0fe", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.265625, 3.1015625, -1.42578125, -1.296875], "student_probs": [0.004541118163615465, 0.972977340221405, 0.010517253540456295, 0.011964253149926662], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "99b3fd429034001b5c03fab18c25d916825f06c2961670ee9cb43da4d716d2b4:action", "state_id": "99d903ca6f9ac93f43a5818c04a26e15e69113d9d0518e1c715d3c3624cd4544", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3671875, 3.078125, -1.625, -1.50390625], "student_probs": [0.0042168982326984406, 0.9769273996353149, 0.008857701905071735, 0.009997961111366749], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d82772fa43dc80dc070faabcf4aaf8e9ccb6084717dbf2c805eb36ff1417535d:action", "state_id": "5997edcdc4e957d723ce8da967be5b6e08129ae67233adb8affaa128c4de4f32", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.359375, 3.078125, -1.6328125, -1.5390625], "student_probs": [0.004251592792570591, 0.9773000478744507, 0.008792122825980186, 0.00965625885874033], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "44d48ef302070f64974a5c09ca267e3ab1302e03368b526f0183812d904f1b7a:action", "state_id": "b69c08f672ae0c047e1cfbee18eef0ab33369998d33e715ac4ce6c5449d140ab", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.90234375, 3.111328125, -1.60546875, -1.4296875], "student_probs": [0.006476429291069508, 0.9744188189506531, 0.008714988827705383, 0.010389811359345913], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5fcbfe7949a4154b218b7788c022e07375af076ebb749175125706671585a8ac:action", "state_id": "4abbabcf074775e5fee4168e0deaf62f8dc5e0327244af3fbf3f6beaa93e65ba", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.05859375, 3.14453125, -1.73046875, -1.546875], "student_probs": [0.005379348061978817, 0.9781785607337952, 0.007468485273420811, 0.008973591960966587], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3db23b6c1773ea94089443b7aea78762ad1a7d3f4d2da3dcd246faeef7555c76:action", "state_id": "2a87237fe3cf26f728586a944d40061ec0cc2af7811ec4c342d3095ca0a80a62", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9453125, 3.216796875, -1.51953125, -1.25390625], "student_probs": [0.005584734957665205, 0.9747161269187927, 0.008549032732844353, 0.011150040663778782], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "61e34e2a4ece3504d9d010d0f6abad68c1fbcaa3d179c71715818b351d9aa50b:action", "state_id": "4943dbe11f23b6ffcb6b30271affb7e6f1217dc6a96d2677f4bb3afea9e479a9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.04296875, 3.171875, -1.546875, -1.46875], "student_probs": [0.0053078243508934975, 0.9765499234199524, 0.008717006072402, 0.009425331838428974], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6bede97165afad47e35844ae2ddca928c3522aff76e12de531cf7c0c89879ccf:action", "state_id": "0efa7d33ec1ca89b137624236601df86b0f0d1b01783159d6b85176838a60a9b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.21875, 3.091796875, -1.54296875, -1.43359375], "student_probs": [0.0048165088519454, 0.9751548767089844, 0.009467176161706448, 0.010561399161815643], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e59d2520ed3e7ba48567df87b41113e2fb1b7b26ac152f1ab7b688bad12a138a:action", "state_id": "2911fbf4becc2593e78059ada7a032b6f2058e730ffa8888e828b2e130134359", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.33203125, 3.1171875, -1.60546875, -1.5234375], "student_probs": [0.004203639458864927, 0.9776672720909119, 0.00869295746088028, 0.009436115622520447], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4a7f45121c40af7aa5ad3d9138919e18fdab7233f01dbbe15eee330d4b8c69e1:action", "state_id": "2950f1941abc1e7288a51f942caf5864bfa5aa69c3587cfe905d2d491892a15f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.375, 3.17578125, -1.65625, -1.51953125], "student_probs": [0.003804553532972932, 0.9794389009475708, 0.007806437090039253, 0.008950123563408852], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a5f595095bd90ae48efa4c2b8eb366088af627e9041a34da1325f47c5ed85de4:action", "state_id": "4db4defc0d2c47bc394fdb3f072aba1c5cf8e97a810fe15fa8643b2fa8cf2e3f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.25390625, 3.234375, -1.32421875, -1.265625], "student_probs": [0.0040312581695616245, 0.9749242663383484, 0.010214068926870823, 0.010830430313944817], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0e201004c439621add05db2b2c3892dfb925f0e44f9befda48463b209ace3c09:action", "state_id": "5cf4e971587d2b9260b6972d2d1e58d5084231915fd5df98e6b8414aa35d11d6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.25390625, 3.216796875, -1.375, -1.2890625], "student_probs": [0.00410408154129982, 0.975241482257843, 0.009883712977170944, 0.010770659893751144], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2504bb5d9735f3dc161690e40801807647b7bd3a3792466cd2dd40cda95965ae:action", "state_id": "308de15307aa73fe5316aa4bc28341fb08ebe3dbcd570eac0c3887070f719c8b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.10546875, 3.244140625, -1.30078125, -1.171875], "student_probs": [0.004623087588697672, 0.9732803702354431, 0.01033721398562193, 0.011759443208575249], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "df115b279b9a31c36dd1d318a319b7dd4752b88cb18109f5a698d413e4d0adb6:action", "state_id": "1eb4d338badc2e23abb47b60a7b3cc81a344d586f07da7a808e9bf377e038579", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.01171875, 3.28515625, -1.296875, -1.06640625], "student_probs": [0.004870227538049221, 0.9726415276527405, 0.009954098612070084, 0.012534101493656635], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "850c8bcaabcf80724350a09198a50d987d96ddc4413f17888f4e47a24aa91821:action", "state_id": "40c8c6be1687f46bade40108e184f5fa1bea30a7474fb75e13845ff4263bf74c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.90234375, 3.263671875, -1.28125, -0.921875], "student_probs": [0.005532748997211456, 0.9694223999977112, 0.010296237654983997, 0.014748679473996162], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "600fd9d3c11183e4f1ba3fccdf5ecb4bf55c1ad7ae62db1120c3f7a363d2179b:action", "state_id": "2e314f74ff7f579035d332f8ed56e7f7a5ccdef2c7ec900cb5d063c742d0a193", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.95703125, 3.126953125, -1.30859375, -0.943359375], "student_probs": [0.005985007621347904, 0.9660754799842834, 0.011446626856923103, 0.01649289019405842], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a302584a11a5e1b01ee2fedc438e8d07a2f091192d3b37d5b4b1a6ded23122b3:action", "state_id": "59df6237e543e4efc19dfc6a1b2691a075d3a503bac46d6085012a41d3ef58cd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9375, 3.11328125, -1.23828125, -0.92578125], "student_probs": [0.006176388822495937, 0.9644085764884949, 0.012428006157279015, 0.016987070441246033], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "19a9f7be187777b37a0e884f49e443a5a19b66f0699e7094a4ae7be3731d6b09:action", "state_id": "aff92baec5bf86b80113a823dad53f8b5f3004192771ae2255dc7126bbb98943", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9609375, 3.111328125, -1.1875, -0.8359375], "student_probs": [0.006031990051269531, 0.9623156785964966, 0.013072546571493149, 0.018579836934804916], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1d779bd5200c8b46ca8b07fa7038c49a70f6c9c05acca7e5a6caa230a2c3349e:action", "state_id": "9b5c2a21a84b33fe7f10c463bd0c3eddefa914fb3eab3e10c9e241fb2d6e52d3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9140625, 3.115234375, -0.98046875, -0.65771484375], "student_probs": [0.006254624109715223, 0.9558661580085754, 0.015909474343061447, 0.02196979708969593], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5a83c6c1cf6d3c99f400df28093a6161c47bc9f8526ccc6622f75e6626936243:action", "state_id": "6b24f7c2adfaae5213e46f315e44bb39acd04c2a23db960c04cc0264466e0477", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8125, 3.12890625, -0.7314453125, -0.30859375], "student_probs": [0.006737911142408848, 0.9430847764015198, 0.019861925393342972, 0.03031541220843792], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b4b30adb3005b8672a0fa01420552fd33737d96a2bdb28ba816c12b6923586ba:action", "state_id": "334707eee96d228f9d215dd2d9c05e6ed9c5b50e9424f563f5ed2f27e398be8a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.70703125, 3.017578125, -0.2734375, 0.08984375], "student_probs": [0.008070308715105057, 0.9094146490097046, 0.033844806253910065, 0.04867019131779671], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1c60e1920ad766d765854797736d88526f59b33bf5a98d521b92021f2de6b1ac:action", "state_id": "2b45e592109b09d0ab6031e53d2423a799f3ae4533d99fe53bc2f16b32f90202", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.48828125, 3.017578125, -0.0625, 0.3125], "student_probs": [0.00982688833028078, 0.8897866606712341, 0.04089073836803436, 0.059495676308870316], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ddd8e44219766934e81065f9c513ecaf5a7ad14300597a54bcb61ba889760917:action", "state_id": "d5457162a7447980bbe73065df33a1b1a127a863dd82fa60ae982052f65e5944", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3203125, 3.095703125, 0.04296875, 0.421875], "student_probs": [0.010708395391702652, 0.8862895369529724, 0.04185910150408745, 0.06114300712943077], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c1cf736e933fa111e5ea7f4ba361e2977a3b53f294ccff066d0c2b3acd09b490:action", "state_id": "163df2af7b97e6e1650328b6342a2d223320f07b995c26640ccc6d2963f6e3e5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.52734375, 2.87109375, -0.1328125, 0.27734375], "student_probs": [0.010818429291248322, 0.879794716835022, 0.04363162815570831, 0.06575518846511841], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1c184529db68b110cac1de4c1a2273842504d4c27354c387f3aca0a3d7c8e4d7:action", "state_id": "a10a42881b8340fed00c31a4eee1843b4e8b82ba4c936e506f4ad30b1328b019", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4140625, 2.888671875, -0.087890625, 0.3203125], "student_probs": [0.011857672594487667, 0.8763009309768677, 0.044663071632385254, 0.06717829406261444], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b46848ab63aef7f592721e781bf4db07ce86851b160159a08426cd3c0d5ff7b5:action", "state_id": "d60ddcf85eae164954c75d29871cd72040f69233b3cc74c686ceb6c808c54a5e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.35546875, 2.884765625, 0.04296875, 0.5], "student_probs": [0.012365879490971565, 0.8584903478622437, 0.05006782338023186, 0.07907602936029434], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7b4067f97874ae86f8d91e06a2027d4e9952c4fac12fddbd3d572eb88a9dd045:action", "state_id": "52fb87b1dfd825e6662dd211387476d92673c807b2ca67b13a680da1390bf521", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4921875, 2.796875, -0.12109375, 0.37109375], "student_probs": [0.011864844709634781, 0.8649245500564575, 0.046743422746658325, 0.0764671340584755], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e2b124dace8e6c7c4ffbff31e45ebacebaf992193e4a408318590707ac6b2542:action", "state_id": "3ba3751f1b69964f607eec2aacd85a0b43333fc054af5c203566b3fa717b02c7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.8046875, -4.140625, -0.984375, -0.900390625], "student_probs": [0.07066446542739868, 0.018578505143523216, 0.4362673759460449, 0.47448959946632385], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1a092230dc09d5493a13f62e5c084488f1d24a647fb5f6a8ae3f93fa2be7f32c:action", "state_id": "727666f1fd53a09f2ba1c9092501bfe0c1884284c4741c5e8d5fec6cd68dc4f2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5078125, -3.65625, -1.11328125, -0.951171875], "student_probs": [0.23014132678508759, 0.026849739253520966, 0.3414580523967743, 0.40155085921287537], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "554e71d9869b9a1aca92cac4fa47675d4e2048e5cb73f34e9c1371c936e2f6df:action", "state_id": "7a765b8e390920e5ed7a92c8f54aa1d4b78f8f38ff06a5ef7b5e44250c520770", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-3.05859375, -2.98828125, -1.4609375, -2.09375], "student_probs": [0.10374888777732849, 0.11130630224943161, 0.5126686096191406, 0.27227625250816345], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f7b856284d2075e2a7780ddc98f3a17be5bd6a2f78b055542f4fa595c5a7fa2b:action", "state_id": "bf59a20a10471999e054b530f83aec3dd7fb1559e5d0dc621a2d6566487bafea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0400390625, -0.4658203125, -0.2265625, 0.02734375], "student_probs": [0.1259566694498062, 0.22366662323474884, 0.28412505984306335, 0.366251677274704], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f3fc874a4f6de3c1d8b81feef63b94a5a7fde7314c7b07161284e611f1438db3:action", "state_id": "c120ad35840646c8d2795d97926c88b5dd94adb7e58eb291282d41221a4df356", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.935546875, -0.0234375, -0.03515625, 0.25390625], "student_probs": [0.10827881842851639, 0.2695675194263458, 0.2664269506931305, 0.3557266592979431], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "06fba4c1d6bdd78d467a1cc23c7c73708c28d0b18d684e7171578e83e7463283:action", "state_id": "8ae23832529ee523ecadc7d1e560a9a9b107f5ad27e71b6bf16deb38471b6fb4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.53515625, -3.5703125, -0.89453125, -0.736328125], "student_probs": [0.1904304474592209, 0.024881655350327492, 0.3613734841346741, 0.4233143925666809], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "53b9afe9a82a15471c9ffefefef2494024d3ef321c522c20acbf9bd8909c71a4:action", "state_id": "730be4186f2180e9d57f02d811046b7e3f410478ab90d63946a05ffc8a3860b9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.15234375, -3.84375, -0.974609375, -0.724609375], "student_probs": [0.2634302079677582, 0.017856759950518608, 0.31466948986053467, 0.40404361486434937], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b78a923b61e43c0d1bf8e984c67f3df192dffd472d34421202a0c71bdc2d2a4e:action", "state_id": "51ed66458add52952df90d746a30374e2b5dbaf91fa1da8dea7d69c7a88eb0fe", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.33984375, 2.287109375, -1.5703125, -1.29296875], "student_probs": [0.024727847427129745, 0.9297196865081787, 0.019637897610664368, 0.02591456100344658], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ce0c8935cfd98e9dcf28df1c895b45e25216a734fe1e3a85c5b0b6b6a739c667:action", "state_id": "a1da3d52c8587cc0ebe41bb5f484636239fa61fb699da1e3c5e6cfd8cafd555a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.865234375, 0.53515625, 0.1484375, 0.662109375], "student_probs": [0.33075809478759766, 0.23777125775814056, 0.1615137755870819, 0.26995688676834106], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "83d8fb07ec071d53ec32146698f76256567982b6b9bb45933f68d621bfa246ee:action", "state_id": "1eea23b153f258b985a337de2a3c0b6ddfe4acc10c66a52630b9c89c76cf6ffd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.88671875, 0.767578125, 0.5625, 1.544921875], "student_probs": [0.4342042803764343, 0.141793891787529, 0.1155029833316803, 0.30849888920783997], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f1374d6e7328cbee2a6c6463dd3264898c0eb8da99c1dcdcb241b797b66046ca:action", "state_id": "f6c25dc716b7ac05a5ce43562d0df82b53bbccf17000a79739d9312d88fc7593", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.484375, 0.626953125, 0.55078125, 1.66796875], "student_probs": [0.33124426007270813, 0.14053185284137726, 0.13022480905056, 0.397999107837677], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "905812fd1451a01ba2f36b65948cb00a5601b5118b82a4f4a39b6f42dbeae371:action", "state_id": "b5db2ab737ecc226363e028e3800b3dfb591c00bd1ad821c72bb4dce207410e8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.2734375, -2.8828125, -0.3994140625, 0.03125], "student_probs": [0.42776253819465637, 0.018216324970126152, 0.21826645731925964, 0.3357546627521515], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "488e58131d313f2196302006c6e031b12baeb2eb4f95ca19bdf28a8d809808b3:action", "state_id": "528cb93212e3b51c5970f640508bdabd73a2e12b7ad9bd6d8144c9d955748759", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.3203125, -3.171875, -0.224609375, -0.4599609375], "student_probs": [0.3302673101425171, 0.01907426305115223, 0.36343684792518616, 0.2872215807437897], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3292176ec630714b5493adc08f574b4b68132edbe53ec44851b7ef9b177054bc:action", "state_id": "93847a77631450cbf8c502ebb2db6434833bab089d61808bc5f097d6cd469327", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.884765625, -2.63671875, 0.87109375, -1.4140625], "student_probs": [0.13243499398231506, 0.022968847304582596, 0.7665894031524658, 0.07800672948360443], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "48c62341a54fbe7c21b812e73ea7598c04574d8133414526568c90769cc28337:action", "state_id": "dac90d74719e3ae0aa544c45e175956a2a51b384f90483bd9cab8af15cdf44c3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.625, -0.232421875, -0.32421875, 0.35546875], "student_probs": [0.1539074033498764, 0.22790510952472687, 0.20791563391685486, 0.4102718234062195], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3ce70a5a52da8c014d30507002309dfe1949eae68e11c72254108a71d95d92d4:action", "state_id": "5ccacdfc87ad7121f70dc6403eb9e5ebb309e1ab4c17a7adf646252e32a1e9cf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3828125, -3.21484375, -1.05859375, -0.546875], "student_probs": [0.206184983253479, 0.033007752150297165, 0.2851434648036957, 0.4756637513637543], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c2139525568b5f4db3d4fadfe94f8ffee84ef2156361507db49414f4376b8b2a:action", "state_id": "57415a260176d1db7dd776b1f08f858f2f92c514d8a1d17109fe84678b144489", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.837890625, -3.15625, -0.908203125, -0.49169921875], "student_probs": [0.2903423011302948, 0.028579827398061752, 0.27062878012657166, 0.4104491174221039], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "be37abf02f80b375aa8e97e72da8ae7b30c36ff54b475fd069691b776840386c:action", "state_id": "3b353c964a07e48f2d4d37f94b69153db350edd8f13f81375d497dadc5aa4412", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.84765625, -2.75, -0.71484375, -1.87890625], "student_probs": [0.075896717607975, 0.08368249237537384, 0.6404595375061035, 0.19996123015880585], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "626b71989a61682ce51f8e29cb1247c72d0235869172948ead2e1ca5a24375f7:action", "state_id": "a17deee6071bf6b49b265568a9abf0c0695e777684e14a9d334dcbd3d4739fce", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.79901123046875, -0.15625, 0.03125, 0.37109375], "student_probs": [0.11879344284534454, 0.2259124219417572, 0.2725023925304413, 0.38279178738594055], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ee72709acbae9af6a07dbc7eea25ee173c54b142a435006eda55955869f36a88:action", "state_id": "a15f930492121084e68b4613a1a5719d8a31bd12ce6e68d8d97240558a2d3927", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6484375, -3.8515625, -0.853515625, -0.73046875], "student_probs": [0.17155824601650238, 0.018949884921312332, 0.37987592816352844, 0.42961591482162476], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9077869268ddc2560bf685a8bd95da1596967e7acb2b82d7c90e9b2dd9a3e02c:action", "state_id": "1af447b9373d3b7b45f5188229e9a53b0c87710346b64fd4d4c703c5c44b96ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8671875, -3.7734375, -0.8232421875, -0.66552734375], "student_probs": [0.300929456949234, 0.016454940661787987, 0.3144487738609314, 0.3681667745113373], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "31ed72b686bce13ec00969be78a79a4a20fc9c5a90bc5eb48f4943622269969c:action", "state_id": "12687c7ee4c50963520e64881af9c273151cd6c70b9040f3f33630ebaac35b33", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8125, 2.2265625, -0.517578125, -0.6953125], "student_probs": [0.015508672222495079, 0.8804753422737122, 0.05661768093705177, 0.04739832878112793], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "243d32ca643b3f0c54e6dc95a8972129a7819e893ca56c02e220a367f0fdb9c3:action", "state_id": "0b95b23d78702562734f5b7f7c5044ee62b1e7a517b4df809f00c0efefaccf6f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.00390625, 0.6484375, -1.083984375, 0.671875], "student_probs": [0.19138123095035553, 0.3674587905406952, 0.06498713791370392, 0.37617284059524536], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "75ecc743636a3248bbfbde761b88bec68afc64ddb7e1801bae9f82bd5fb3625f:action", "state_id": "b5a1748321f6eca73a2308b3ea9234904d8b0856f31609ea94abd296daa5b76c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.888671875, -4.140625, -1.234375, -0.8505859375], "student_probs": [0.35903531312942505, 0.013894145376980305, 0.2540974020957947, 0.37297323346138], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5821d5efa2404853ff6212b4d47e7c15783e0288a5a3db54545f398fb3f70bed:action", "state_id": "f04885292691b99b1668c7f5015260f9a1976a2041d0bc24c6fd1cc8c84699e6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.53564453125, -3.21875, -0.875, -0.7734375], "student_probs": [0.3892647624015808, 0.026606464758515358, 0.27724575996398926, 0.3068830668926239], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4065280e388c1ebee3d29b9137b117b4f6ebfaaf26f4a73a8580573b891ca8a1:action", "state_id": "5811b3cca5e8b04e595a54326c811c805884ef46b51d2be4d8bb3ee376ebb770", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.77734375, 2.58984375, -0.14453125, -0.6201171875], "student_probs": [0.01134803518652916, 0.8944706916809082, 0.058082081377506256, 0.03609921783208847], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "13afd51e3aed3a9fea1579ccf4a335d0ba28ffdfb518d59454fdc5a5e56dfd12:action", "state_id": "9240c447f858e9b509ec2efefabe3088b496c953156f5d8dd9765acf96e8145a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.955078125, 0.34375, -0.125, 0.2734375], "student_probs": [0.09638877213001251, 0.35326477885246277, 0.22106745839118958, 0.32927897572517395], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c5d18638a49cc9aad8b6ae560d5673c7a99dadd67086d87fb2262e4c5590df46:action", "state_id": "082dfd836ebbd795ae024b389382ad3e4ca291cc10cd505b4e59a965772b863e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.6640625, -4.234375, -1.21875, -1.2578125], "student_probs": [0.104912169277668, 0.021819651126861572, 0.44516101479530334, 0.4281071722507477], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d484aa163d263acab0b6be4cd0a843277033216c6df2c60e9d776233eb0a9601:action", "state_id": "0dbb423342a59ec2e5261c4453e5cb7280979d910825497ae88c731f0ee28538", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6640625, -3.703125, -1.0546875, -0.87109375], "student_probs": [0.1930733323097229, 0.025128623470664024, 0.35511618852615356, 0.42668190598487854], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5c5b59dc50dffcb3e7d15715f0adc963a18d981a457bd9a094ed9b64159e788b:action", "state_id": "751c80800b45193847201f243047a5af4144e6ffef88450c393759a7193131b0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.28125, 2.2109375, -0.45703125, -1.099609375], "student_probs": [0.026783064007759094, 0.8800311088562012, 0.061068031936883926, 0.032117802649736404], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ca963e869de744f0d26607d5b23e7dc179feb7732bde4f0616865b56a2535474:action", "state_id": "f7a73017ac00cf86e357166694879b5807ee1199704a1bedacc5d2befb6f328b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.265625, 0.892578125, 1.150390625, 0.46484375], "student_probs": [0.15349750220775604, 0.2873317301273346, 0.3718348741531372, 0.18733584880828857], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "eb2c8307d96541787d1c7de1b54c531fd49af6c6cde620ff69169e1f248cf8b0:action", "state_id": "14c087fea1ca82242133ab69c831efdfb95ae00482ebab393e62c2eaf94d2e99", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.2890625, -2.09765625, -0.046875, -0.77734375], "student_probs": [0.327697217464447, 0.05370447039604187, 0.41749706864356995, 0.20110130310058594], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4d3756c76eca0e73700ed171bdee394f08d859d97764a2ec1e50fc813618a12f:action", "state_id": "bf18f7cea41735371b47fc98db8e59e5f842f62c71f9b7ad500233b4864bae23", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.265625, -1.703125, -0.6171875, -0.265625], "student_probs": [0.34000784158706665, 0.08075893670320511, 0.23922540247440338, 0.34000784158706665], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "262bb7778362b11222c104790e5d2064d9eb604bc6713e459d323e6134774087:action", "state_id": "181e19c1209e201a998078fc15853668a9b1d34444368900d5e3a81af5f9add3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.015625, 2.552734375, -0.530029296875, -0.947265625], "student_probs": [0.009549817070364952, 0.92046719789505, 0.04218723624944687, 0.02779570035636425], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8f03f3a6344f2b60d12d88f88c420d74b4e493e73f0daf547ea03627fe0571cc:action", "state_id": "b8ce28b929a3053f39d92959f6047dde5a57a85723fd795716a7c8b19abf0359", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.095703125, 0.30859375, -0.18359375, 0.2421875], "student_probs": [0.08792579919099808, 0.3580920696258545, 0.21889728307724, 0.3350848853588104], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d20ae4558565890f19f9d9a2569a1a39b23db7df675d2f238c1b4a826e1144f1:action", "state_id": "0de7171d29ddd1648c59a231250f4ac83c63587cdf69cd4778294303dd10ebc5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.703125, 1.1328125, -0.52978515625, -0.0703125], "student_probs": [0.037882525473833084, 0.6457597017288208, 0.12246555835008621, 0.193892240524292], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c4686d9f93de7f4fa1c6c99a0131c208dd9db8f51ed6c67c7502aae6c61989f8:action", "state_id": "bc9e8c0d5b00c8d104c6c475e4795d3c16d1da086dbc486a7bf6f116cd42a56a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.71484375, 1.513671875, -0.6595458984375, -0.208984375], "student_probs": [0.02974153310060501, 0.7507405281066895, 0.08544238656759262, 0.1340755969285965], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a0893472e1f25b7753344a8e1d56f26f0ebd75c0bc7608f2afc0f3ee588ad9ab:action", "state_id": "2f167b2a6984a802e881e5da054c9f1e160b497c401f1e07a45f341d1319139d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.06640625, 3.041015625, -1.30859375, -1.083984375], "student_probs": [0.0058463020250201225, 0.9660651087760925, 0.012473693117499352, 0.015614988282322884], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "17b2edc4cd662041284790cd4439b6a8c55f1dd6da6f77e45a1e64b917124a30:action", "state_id": "8ff04fd518d1b251465813e4a6284517a39d69abd232e58f93abd4a923ce2891", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.21875, 3.044921875, -1.4453125, -1.32421875], "student_probs": [0.005030105356127024, 0.9717639088630676, 0.010901261121034622, 0.012304588221013546], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ff021d5fe68176709b6325abcf55d48f79ba33f8e432f6b8dd3bbbb8f8a8e7ac:action", "state_id": "67ff9aba49a749311688f1a3de595c1b3d48833e631d0981427a13d9370c525e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.296875, 3.052734375, -1.5078125, -1.4921875], "student_probs": [0.004630414769053459, 0.9748229384422302, 0.010193079710006714, 0.01035359688103199], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f49a7a1e9cc16c21d1607e431bfe5cb94e301f4ae1d6189d5f8b513058191969:action", "state_id": "64875a84c0d0043883651a3782c5b108a04abffb02f84a74b0abecddaf7f3286", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3671875, 3.078125, -1.5546875, -1.53125], "student_probs": [0.004215315915644169, 0.9765607714653015, 0.00949936080724001, 0.00972463097423315], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "95877e27f86614b13933f9e5a318fa57e0bc78d1b1da08853f2175a27b42ffc9:action", "state_id": "a6468e619e2b870ecc2b345fdcd5a882e7151bc174f7431fdb312b01029f9816", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.36328125, 3.0625, -1.5, -1.4609375], "student_probs": [0.004291383549571037, 0.9749541878700256, 0.010174560360610485, 0.010579868219792843], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e1e010aa6036f11e76e4c6cb83b4caf75193b3494b7d980b62918e10c615d2ac:action", "state_id": "76a9a7ba1730eba24de3fd128b0f3e42b63398e1c48471a88aa85f3ebc2a6d64", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.87890625, 3.111328125, -1.5703125, -1.390625], "student_probs": [0.006624188274145126, 0.9735627174377441, 0.009018893353641033, 0.010794201865792274], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a5f0fe08b1efe35b8773b2687e1ef3c5b43f988e0f178f43ad9dbc0aafd0e123:action", "state_id": "654af4d4041e72dbab6e0967ea276b0b1f2ffdde7096e4c5c4ffaab5e4f21273", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.05859375, 3.091796875, -1.80078125, -1.59765625], "student_probs": [0.0056696245446801186, 0.978003740310669, 0.007337038870900869, 0.008989527821540833], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ce9566bbbdd7a19ef08eddf5d0576f554d49925ed14b50321d055d3843650135:action", "state_id": "950c7fa3f944cab9d51d3e5326afbe66ccdcf526905229ea6a4002dba69662ec", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.02734375, 3.05859375, -1.59375, -1.46875], "student_probs": [0.00602327985689044, 0.9741540551185608, 0.009292668662965298, 0.010529972612857819], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "79f3cc81f2345d175bd858b63f0a18091e33aff665292353619a1875ea88b82d:action", "state_id": "a5c5d8d5caf4af238648396f7079bfcfc3d6954e6c96b2dc4471b269a7181622", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.72265625, 3.107421875, -1.33984375, -1.2265625], "student_probs": [0.00773219158872962, 0.9682307839393616, 0.011338510550558567, 0.012698528356850147], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1d2f4cfcc483876b2ea180461fbe7d9be9029b5f9f2fcaa82516eaa5758b23a6:action", "state_id": "3de7cd97a95fefe44822e81231032a01bcb648796abd13124ce602ab38b73af5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.48046875, 3.265625, -1.06640625, -0.77392578125], "student_probs": [0.008356062695384026, 0.9620640873908997, 0.012642319314181805, 0.016937503591179848], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3045932eb5b78f1c2a845594545fbe33957e10733bd7f28005b03ee961ffeee5:action", "state_id": "0e07891eec823c185dd5441d4f5c9faa5a83abf5a91a1ec485c21afe5729b94d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.955078125, 3.44921875, -0.85302734375, -0.291015625], "student_probs": [0.01164799090474844, 0.9528244137763977, 0.012899448163807392, 0.022628184407949448], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d772099d4ceca241c018e6b1cf8d36e30f912622a57398047f29dbe332ee5613:action", "state_id": "34d31dd6ec2301d59a7c850a4279f607307ba80fb424870fed1769fcb8885ab6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.00390625, 4.1455078125, -0.0390625, 0.26171875], "student_probs": [0.015115895308554173, 0.9508424401283264, 0.014480140060186386, 0.019561421126127243], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9d4ee44f06edefb5d22d334dd60a2e52fa8b323b965df57239a5044d310ddd95:action", "state_id": "d376e6dea3d46e9088439039de498d9c802e5347b0367a16ddeff890b2ef4e8f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.0234375, 4.0830078125, -0.13671875, 0.265625], "student_probs": [0.015635129064321518, 0.9495285153388977, 0.013960598967969418, 0.020875636488199234], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "89f84f3107f63bb15fdc0c885c7dc3285b548d2b40d3e72879077d9c3bb65639:action", "state_id": "fdb13470dce72c04ca548735e7686efd6fb00eb648400dbed6ccb369f1b92355", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.1015625, 4.044677734375, -0.140625, 0.21875], "student_probs": [0.015029638074338436, 0.9498122930526733, 0.014453861862421036, 0.020704202353954315], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ab8e677f95357f285939bce660fd0d1f51884ee0234e06925621e0ba4f5f7cb9:action", "state_id": "2ab71724db23a20c337bd8a5f9fca3a8bf0f7ec9825264a5a08476a23e23fb86", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.0078125, 4.09326171875, 0.08203125, 0.3203125], "student_probs": [0.01565251313149929, 0.9454922676086426, 0.017123902216553688, 0.021731361746788025], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e3159858ddb88ec36a23560d27c7e5bf8698fcb9327f39ea3a17fbba2517eea3:action", "state_id": "cbfb00a6e300f467692c6d88f025b563c0d15c047f25e74a8e67d9a7a307ef70", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.2109375, 4.102294921875, -0.02734375, 0.29296875], "student_probs": [0.012732655741274357, 0.9508938789367676, 0.015298637561500072, 0.02107476443052292], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "966cd8dc872bd150776a8f69ab3ff914ca4481160f1ca07e672646366edb8056:action", "state_id": "a481cf1019f9c0cb96aa00bb9318a51030fa64e5def2c8800d1ef818da12aca9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.1875, 4.11181640625, -0.03515625, 0.3125], "student_probs": [0.012909438461065292, 0.950772762298584, 0.015033820644021034, 0.021284066140651703], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b13f669ab4b7d33ce820bdd96417aa5fd6f07eff20d80c28a5c600a733f524c7:action", "state_id": "30980d7c33b64d86293d62effcae9bc4f860e5600c3399bef56d3e5e06061c8e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.05859375, 4.12744140625, 0.2109375, 0.3828125], "student_probs": [0.014362495392560959, 0.9444997906684875, 0.01880553923547268, 0.02233213372528553], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7831da26a544a7b8fb79bae28a72e041eace52c61b157148f99aa1d5b69e32e0:action", "state_id": "4247cf4c9ccb53fc90d66e0257b98776fc4c53a7f52c21943027ce368faee566", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0703125, 4.224609375, 0.41015625, 0.33203125], "student_probs": [0.01483436580747366, 0.9450552463531494, 0.02083825133740902, 0.019272230565547943], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "23212cb2b551c1516c74c8e3d52e79ab94128c827c249e6276c6315bc5a6cf8d:action", "state_id": "42ac712d22254a1cdc13e88dc9794fc1953da4619780b79db34fc9bef571caa2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.10546875, 4.203125, 0.52734375, 0.41015625], "student_probs": [0.015605480410158634, 0.939434826374054, 0.023795517161488533, 0.021164171397686005], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7eca966f46e85418c4e0fcf037c3cc461c78841d2f0ca03f0e43662e4859bb1d:action", "state_id": "99bb69a6a9143a3b43524b22b258766149e5ac2726ac6f12ac62181167458d77", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.171875, 4.2421875, 0.732421875, 0.4140625], "student_probs": [0.015974203124642372, 0.9356932640075684, 0.027980899438261986, 0.02035166509449482], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9bace8e0d77c1ca8d1c35c56ed3c0ff875d5d845b98a0d0bcd6fc6054c3dde5d:action", "state_id": "0e638a14991e8646d36a86acfc0d4373adca6c7dfaea77dde4d7085d0b102767", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.15625, 4.2392578125, 0.76171875, 0.39453125], "student_probs": [0.015766700729727745, 0.935338020324707, 0.02888634242117405, 0.020008984953165054], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f0ecf868051570772b71851a6751c1699fb4cfec2743cbc6aab3e3376e15f2c0:action", "state_id": "66110404bdcdb050829aa9c74aea5bc92c6652d9c184f5c890967c0a79b2fe81", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.11328125, 4.25390625, 0.658203125, 0.375], "student_probs": [0.014955347403883934, 0.939825177192688, 0.025790102779865265, 0.01942940428853035], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "37ea7ddf02dc20c4d165e149ae1fa455d69a64fdf1ddf24207c0654f5cd5b193:action", "state_id": "c239c87353ec41b793cae1d61c686fecc574fb1165a8b56266ede7e932068aec", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.1953125, 4.291015625, 0.728515625, 0.41796875], "student_probs": [0.015616378746926785, 0.9382565021514893, 0.026616288349032402, 0.019510962069034576], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0b2604c2ea1078e320dee3d295d25a9735297ec01e5b34f297b5ddd6773f6615:action", "state_id": "d7f68cd8f101e6857178c3a58087e44348013a1c8c150b1a7fdfc27dde51e132", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.70703125, 3.04296875, -1.26953125, -1.119140625], "student_probs": [0.008337968029081821, 0.9637380242347717, 0.012914096005260944, 0.01500990241765976], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "33b00fc55261c3834fbb1417679db30f83d45e30bd18155d558fd9c641130aa9:action", "state_id": "d944f0e5452c5ce3a5bcdcf147769ca77d36364d226059a3c54eff042e712c7c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9765625, 2.814453125, -1.35546875, -1.2421875], "student_probs": [0.007976469583809376, 0.9605552554130554, 0.014843909069895744, 0.016624389216303825], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e5bad8ee881330d6475533c18483ed141643c0e7c9c5fc687af353609a91ef15:action", "state_id": "c40394417e74300a9e8a0baff3af86ee87c9a44a9408869ef04858b47c042daa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.98046875, 2.90625, -1.296875, -1.15234375], "student_probs": [0.007257523015141487, 0.961752712726593, 0.01437703799456358, 0.016612635925412178], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "996f7f50c9e168d0f398a7e64fc458845f1d4d08e9e8112de747d4b9864aca6a:action", "state_id": "db1e075dcf09cc41d61f363dc04922287c4d067e3703d094a7c34770aaefabe0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8359375, 2.9453125, -1.09765625, -0.7646484375], "student_probs": [0.007983088493347168, 0.9520098567008972, 0.016703305765986443, 0.02330375462770462], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d176a8b8fef6651363a1a40ba59f537c07ee6e8a2eed3554469a88fc00437e77:action", "state_id": "94a2a406da2f1a337d3d00757ff02a287d34e922ac251a8cba0deb6fa4d3ef4d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.76953125, 2.84375, -1.04296875, -0.64013671875], "student_probs": [0.009347877465188503, 0.9424007534980774, 0.019331036135554314, 0.028920304030179977], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1de0f5cc13fde3e913555a6e8eec531db2d84be604fff13644cfd3917722a6d1:action", "state_id": "a981aefe3b5021070a41d0122b166bfe35bf6cc6d2c3eb5bd09eeecb3a0cc22a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.70703125, 2.853515625, -0.794921875, -0.36328125], "student_probs": [0.009712629951536655, 0.9288747906684875, 0.024180255830287933, 0.03723231703042984], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3545928684a49de5cc3a86b45bb8f296f2f1be03b607b5db2645e87b64e465de:action", "state_id": "a6cbba528ccb89574ea5174d5f70c123961a873851b9107094db14c5ba430751", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7265625, 2.853515625, -0.3779296875, 0.140625], "student_probs": [0.009187440387904644, 0.8959776759147644, 0.03539144620299339, 0.05944341793656349], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "38a359a0785d9675900a01c1f78c17f0ae8fecbdecf4b1289f0d155bbed6adbc:action", "state_id": "07f6e0117bf3192952792c78b5f627c37517bab471d8ed8316e66b193fc73de9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6484375, 2.833984375, -0.1796875, 0.296875], "student_probs": [0.009921792894601822, 0.8775689005851746, 0.04309830069541931, 0.06941104680299759], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6bc671c56054e5f9f200a046ae20bdc6afc335c97ddc6d4383df4b4a80ea45d2:action", "state_id": "65976808978c86676a483cba3e4f01aeb9d9eb771ee2ee34df6eaf8eaae3b310", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4140625, 2.876953125, -0.06640625, 0.3359375], "student_probs": [0.011955484747886658, 0.8732358813285828, 0.046009428799152374, 0.06879906356334686], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f7da38eb0920526c330aa16a4412a990951e256b2c6aff5e2f4bb5d9d5900e23:action", "state_id": "1978f56185a023bc5c687809d1848d368986f26bd7af9bb64285ea0f4b6de6b4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.43359375, 2.876953125, -0.3876953125, 0.1953125], "student_probs": [0.01198674738407135, 0.8927874565124512, 0.03411373496055603, 0.06111198291182518], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7044afb8738f683e28862e372772e4f7d0d036172004214dd1089c6d9a280c82:action", "state_id": "325cba83611cf0822149bea1c54f9c2a40149dc45c261ed6cfb1515bcf6c1d2e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.62109375, 2.849609375, -0.63958740234375, -0.08984375], "student_probs": [0.01044816430658102, 0.9133594036102295, 0.02788064442574978, 0.04831182584166527], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "98558ad4c6ceed8ee7606d5fab77181c7959b740efc8356f73ade6cb23aff942:action", "state_id": "ed629d503060d02b213fc1acc7d33bb556db2e7af5cdeec144ba06b68f757eb0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.59375, -0.00390625, 0.0234375, 0.37109375], "student_probs": [0.055322956293821335, 0.2712475061416626, 0.2787667512893677, 0.3946627974510193], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4096f5003d1f3b129cf9243db6a1077a8d041855c9fe1701d9708f1c5cc6eb99:action", "state_id": "b1bd88d80ff01380bf7ea2ac5918aff59598a71c431bb8be03d9d82be2f81a1c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4140625, -0.095703125, 0.25, 0.62890625], "student_probs": [0.05639687553048134, 0.21077118813991547, 0.29781609773635864, 0.43501585721969604], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "83bdfbdb58521e931075034e9d63a376c7f474a6775fb136c15bd39be074c290:action", "state_id": "4a7b2bff2839a401c7a7dbb04e95c4afe3f90f4cd4c0480f86255b38f0cd5987", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.4453125, -3.9296875, -0.869140625, -0.7880859375], "student_probs": [0.08843456953763962, 0.020043160766363144, 0.4277054965496063, 0.46381673216819763], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6835077539c7adc41adf2dad38b0970c4a2130df875c7ed0acef0cb2d103e2dd:action", "state_id": "5d2998502cd69984b806a5ebfe531d48bcfbab8632721aad669df5683d68cb08", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.107421875, -3.2890625, -0.869140625, -0.655517578125], "student_probs": [0.2529580891132355, 0.028547896072268486, 0.32102054357528687, 0.39747345447540283], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9d24460705bd2f71afe88bb4da2e3ffd0f1c36506eac36ac6ae1f8369f3a1a3e:action", "state_id": "85c2c5340311123ef618d8d20ce97ecab54c8b4e7d7ac2ed50e4220fd311a62f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.51171875, 0.08984375, 1.880859375, 0.140625], "student_probs": [0.1592923402786255, 0.10446647554636002, 0.6263327598571777, 0.10990840941667557], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7117c9357f031b76f0be6285a1f77de7bf5bdb34d4f15fb48d8ff997c6e7deb3:action", "state_id": "3b2edd4a2017f055a810526b5609d7dde7c6c6ced5fae842fe471bc6efe26de2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.498046875, 0.078125, 0.140625, 1.158203125], "student_probs": [0.45229682326316833, 0.1093350201845169, 0.1163865253329277, 0.3219817578792572], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1fc74ff66f27fb5b6c107316efbb36f8ee9b9a7328feb1394181b8e6eea588c6:action", "state_id": "dc97f4f04cda735ceb73f712c068dc7dd03a162a3fbf342f3b38be1784e7921e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.31640625, 0.0546875, 0.15234375, 1.435546875], "student_probs": [0.3673890233039856, 0.10403241962194443, 0.1147044450044632, 0.41387414932250977], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3dce5a48d28ebaaa91e003a8d4347255acadc7abf5bd80f39fa55183646aebc4:action", "state_id": "73adb6aa7b67acd2a3e1b04d8f6ebb96c6c702ebcbc651596102509c2f2c58ed", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.23828125, -0.275390625, -0.224609375, 0.765625], "student_probs": [0.1752462238073349, 0.16886213421821594, 0.17765861749649048, 0.47823306918144226], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "22391d2b97279c056c9e0f0fe1f29fa7efc85241cce78880448d476c6792ffa1:action", "state_id": "0cb4eb86d3f99c2100f7b41fa5c2c62ac991de6b8a8ea7f3d1c5137d6301257e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.876953125, -3.234375, -0.935546875, -0.34765625], "student_probs": [0.2677023708820343, 0.025341765955090523, 0.25246739387512207, 0.45448851585388184], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f5824c3be1a39677c1173a19edbd299e614796e1d0f8c9a080dbef757455655d:action", "state_id": "3b7ef41440fbd705c6415d0f3fbb6cf9b0371f503c4f38c008563caa40ab372c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.46484375, -2.89453125, -0.87890625, -0.78515625], "student_probs": [0.19962790608406067, 0.04778767377138138, 0.35866639018058777, 0.393917977809906], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "36b38fddcf934a4262612563c6ca59f9ef49ed66b9dd26767926e776f68297cf:action", "state_id": "9bc61b746a946758b104dc6b6104a77dc1473cde754cd1e630baab0b405c60a3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8046875, 2.619140625, -0.05078125, -0.525146484375], "student_probs": [0.010662445798516273, 0.8894078731536865, 0.061598289757966995, 0.03833137825131416], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "17376ef0ce12555df120bda9800078b12817c9e03de9f672415ed65be111ac93:action", "state_id": "200b404ed4393f40ec5a3682a66921c60676b3cc01c5291505d1b8c7371a6658", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9853515625, 0.296875, -0.05078125, 0.31640625], "student_probs": [0.09236571192741394, 0.33294668793678284, 0.23517411947250366, 0.33951348066329956], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c49b53815188191de0182f06307fbd801995b7f5f173e3521a568e5345aa51d7:action", "state_id": "85aac37376f995fdb9672db0e7cd6f5ea33f19dc31b40058c9008e41c175690e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9248046875, 1.1328125, -0.01171875, 0.4375], "student_probs": [0.06568368524312973, 0.5141257047653198, 0.16368380188941956, 0.2565068304538727], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "acff25f069a51ef44b42abd9e9829060f4ab1d116123e076fe6ebf1a1cf3cb5f:action", "state_id": "6af7240a4c9164d9530e44a431fc9cb68885e0064c2c55229152d198ca8651c4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, 1.580078125, 0.05859375, 0.38671875], "student_probs": [0.04586997628211975, 0.6270618438720703, 0.13694246113300323, 0.19012577831745148], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c0f79f50b8f7132e3d103d600b8cc75faeec2ecd9687e35bf691868e14a39d81:action", "state_id": "8329e7e81d69f7a3dc34f1b9671ef8eef21f7aa382ad4a540a39f40f7d9d93c8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58203125, 3.09375, -0.56640625, -0.345703125], "student_probs": [0.008732068352401257, 0.9370939135551453, 0.024110013619065285, 0.030064059421420097], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ec7b0a80c2a1567a6461522075f543cc3b5aa0f7759b963da81a3c8ca72e98be:action", "state_id": "b87a9b60feb9196aeb410401440a5f89648f3a9a3e2b0658fc2dabb09abca48e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.11328125, 3.0390625, -1.22265625, -1.1171875], "student_probs": [0.005587202496826649, 0.9656702876091003, 0.013614067807793617, 0.015128379687666893], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7f08acd1dec5fc0907042b33fb87834ccfbd30d1af1715c708f9470c7a798193:action", "state_id": "13d95288e2576dfd886a65c815d8cad6429dd9ecb0daa3c76037a22f5ea4b96b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2109375, 3.029296875, -1.546875, -1.48046875], "student_probs": [0.005161742214113474, 0.9740947484970093, 0.010027553886175156, 0.010716053657233715], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "835f71add830811395ab18799d79137d87b04b023cd742437312f78e48aeb32e:action", "state_id": "2b97c7813bd755aa04bd991f2c6b5e2d81da891c48cb0516bba9ead8ab2e61c7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.29296875, 3.052734375, -1.69140625, -1.57421875], "student_probs": [0.004660220351070166, 0.9772728085517883, 0.00850475300103426, 0.00956215150654316], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ce07d3bc9866a16b8ceabf8766d4ad32b02eb19be869dc022b015be4cde39b22:action", "state_id": "0bb6c3c29991d7415e158f6c5f34a974f80801f3e7b8eaf3eb8af7d32282b637", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3203125, 3.052734375, -1.578125, -1.5625], "student_probs": [0.004529956728219986, 0.9762895703315735, 0.009515289217233658, 0.00966513343155384], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f39126ebc97f29966aa4855a61361cedf6a8a83e3b4d1b3476152828cbf1e7d4:action", "state_id": "c023e60b0c2a105a6d3be7aefebaae6286c19912b0edc0a6dd4d9455989d0b54", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.34375, 3.130859375, -1.66796875, -1.60546875], "student_probs": [0.0041048345156013966, 0.9792380928993225, 0.008068331517279148, 0.008588694036006927], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cda300b20fc839bd1151788f9440be58cd042fa0b4f1cf96aac4db19401202bb:action", "state_id": "a4a3c034ecc04b412541fda802b0a5b7e8571539e1028eaacf317dff8b1b414c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.42578125, 3.09375, -1.796875, -1.70703125], "student_probs": [0.003930115140974522, 0.9806346893310547, 0.007371159270405769, 0.008064073510468006], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "791bab846a8b898b31fcf53d1c6f9b9e3b52772bc1674c142394d77cea4a2421:action", "state_id": "1fef0a3095cb0b3751f3e2ed3ee4e6624977b9a87f07cba807f4673e079868fd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3359375, 3.09765625, -1.87109375, -1.69921875], "student_probs": [0.004283524118363857, 0.9808011651039124, 0.006818365305662155, 0.008097008801996708], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b05242b99c5e7eb2c1bcab2bc27650ba710e852eead6a244ee9d3e30e80d31e1:action", "state_id": "51c357d614bad17fe61f5f3e6e1d830582047735abf385079e53a29d41d59a5e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3671875, 3.130859375, -1.734375, -1.6171875], "student_probs": [0.0040126098319888115, 0.9799373745918274, 0.007555337622761726, 0.008494694717228413], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "98956e2366ae577ffb97948a49735adaf5adddee9fa6428b0ffb977698fa5d67:action", "state_id": "8e24d2f9e72b4ad714e88b4479fc617328ee3b132b6e15b035fa3cc67b088013", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.328125, 3.1640625, -1.625, -1.49609375], "student_probs": [0.004030539654195309, 0.978565514087677, 0.008141913451254368, 0.009262105450034142], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5e29a9554f3300213f90b4668091dd8f2a065a911357c118f16c97151276c065:action", "state_id": "1efb55242353250b8a2777785ab3ec9e6f9fb9bcbec52669378bf78a9ee01159", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.3671875, 3.193359375, -1.65625, -1.55859375], "student_probs": [0.003770090639591217, 0.9800915122032166, 0.007675523404031992, 0.008462907746434212], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c8ace3d08b3f38486a4576b7ae65ed99c4647a6b633fcabbe8c8df2131c3aea1:action", "state_id": "f5461d6492de1bf12c6a6eddc3431640c3dc2c48ead93c6d8b0b6aaf5c8f8249", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.34765625, 3.146484375, -1.58203125, -1.5234375], "student_probs": [0.00402102479711175, 0.9781640768051147, 0.008646562695503235, 0.009168333373963833], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8906b36f6e6b9955ffb4d039fdce0f49898cbdafd341a2cf35ff285f35debd72:action", "state_id": "c321726eb1dcc97fb5ed8c0c74d1f14b2847cb9b55237e98732348a63dcb567a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.23828125, 3.15625, -1.3671875, -1.296875], "student_probs": [0.004421804565936327, 0.9736765623092651, 0.010566004551947117, 0.011335667222738266], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7b8adcce6e6be4806b3bef8bc6a75f7131583ab5defa7105211fde014be3a979:action", "state_id": "2461c60570ffdb6dd7f54542d524c07fb93f5bd5b34fa19745db5ee71d0ac1ce", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.25390625, 3.15625, -1.359375, -1.28125], "student_probs": [0.004352411720901728, 0.9734888672828674, 0.010646821931004524, 0.01151195913553238], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aff7bd9110b972680f7b4a32c286f6f2b8d35ac070f2a6731d1db26858dbce94:action", "state_id": "2dac763e2ddf98acdb0d0c2fc9875e799efae90d0aa3f0a533dadbc4712fd166", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.1875, 3.16015625, -1.296875, -1.27734375], "student_probs": [0.0046288445591926575, 0.9725908637046814, 0.011278883554041386, 0.011501340195536613], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e0d41a119ddb681e422ce9f4b94c90a7c762d0fb2b764ec0f552300d61ea6fe4:action", "state_id": "0c5b1a7f44f32d6e469597a48a48d0b41c044be479bc3947e82f3bc3fc205658", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2109375, 3.20703125, -1.16015625, -1.23046875], "student_probs": [0.004311341792345047, 0.9718660116195679, 0.012329939752817154, 0.011492768302559853], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b52736d0b2c0e96d8648362420517185e9799b824c786e5bd0f6f6022dd5243a:action", "state_id": "9c49d432aa1b56d0437cceaa147c7ad2fc1fb4441ebf3114c733130d5fe4a2ce", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0703125, 3.21875, -1.083984375, -1.09375], "student_probs": [0.004890113137662411, 0.9690129160881042, 0.013112206012010574, 0.01298477966338396], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "49c142969843577cd2ad27858f0c352551fdd1d4ec38c487d6aa8e1cde639197:action", "state_id": "c20491280814090d7e40791e55d55cc258af6469d3636034de9e574d984af2e9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.953125, 3.244140625, -0.794921875, -0.63671875], "student_probs": [0.005299657117575407, 0.9580574035644531, 0.01687520183622837, 0.019767681136727333], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c675bbc52b0a30c6e62e1383da2b17a6d8aeb19505be9256e6bff48ec84289be:action", "state_id": "fefeb61007b8c166033cc3734514561831ec216652f9b630c1ca481bf88acbaa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.796875, 3.296875, -0.44921875, -0.234375], "student_probs": [0.0057931020855903625, 0.944275438785553, 0.022294145077466965, 0.027637343853712082], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "13b0b456b90cfc1da1d382e1a7e1c9f1d3fcca51e05f2818f350568b49e34239:action", "state_id": "0f092bbe18f90bf7a6d6c5e4dc218ad2e8d432b08a400983c95f4b7e4f77ab3e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.36328125, 3.337890625, -0.1484375, 0.24609375], "student_probs": [0.008372016251087189, 0.9215587973594666, 0.028211748227477074, 0.04185744747519493], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7596229994e2e25c6dc409737b1e39edb6bb9edce153be596e85daab9d7b688a:action", "state_id": "bb6de07b6474d768f4cc34d2785148bdc583fd7ba959fd25e1d14dcec182fbc2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.49609375, 3.1796875, -0.216796875, 0.171875], "student_probs": [0.008531559258699417, 0.9155759811401367, 0.030663374811410904, 0.04522910714149475], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c1a2dabd6b36154f208239e96ba99b2317dc93cf90fd49243980b5755332503f:action", "state_id": "38ab6e5a2f81d46fb60c706f67a30a1347f57770c482abbe5c80e82ee8bfacff", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5703125, 3.181640625, -0.18359375, 0.1328125], "student_probs": [0.007917466573417187, 0.9169238209724426, 0.0316833071410656, 0.0434754453599453], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6e43a5fb3a90a988929a2e2ab2567fb289b56375a279beeabe05ace72ddedde1:action", "state_id": "85648d7e2b7f59821d5ce8e0d0de093cf89d37e4ca49e3a88adbb62080c2314d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.51171875, 3.1640625, -0.2578125, 0.09375], "student_probs": [0.008561601862311363, 0.9188000559806824, 0.029999883845448494, 0.04263843223452568], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6f9825c348094b21991b2b2b9e8cc022f511438d611db5daeb73e16993055563:action", "state_id": "427463335e729b140cb8e41d7e4ee3011d24d21fc158015797ade5e65b9c9b50", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.40234375, 3.181640625, -0.193359375, 0.140625], "student_probs": [0.009351716376841068, 0.9155676960945129, 0.03132900223135948, 0.04375162348151207], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "930ba5b56d9dbbd053f487d169d6c08c882a69bbd7699bfca74ad16cfdcd4dcc:action", "state_id": "85d565d1084afb9b776467ed1167333a8242038000e309d801a7ab3586101145", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.44921875, 3.1015625, -0.59326171875, -0.1796875], "student_probs": [0.009840661659836769, 0.931973397731781, 0.023161236196756363, 0.035024724900722504], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a06a5a0c1efd6d03bcb8fd3e446b231fa7d99f747917ed8b8119ed4941aa0bcd:action", "state_id": "3b5f94227915229de79d0be3296d39dea058121cf081bc5ff87d3227681646cd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.4453125, -4.0859375, -1.14453125, -1.16015625], "student_probs": [0.11790706217288971, 0.02285732701420784, 0.4329741299152374, 0.4262614846229553], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "437492e47947d0f0e49f528d4e36840d0b8f1e60c4aca247a8e087070ceab618:action", "state_id": "e54074e208436b80c40e415b4f62d3113650309578ed1870aa3e7b304428c5ae", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.125, -3.4296875, -0.994140625, -0.8046875], "student_probs": [0.12324125319719315, 0.03343008831143379, 0.3818405866622925, 0.46148812770843506], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e19503ebb673ee219d3de05df6f02f2c55e6b029fcf1fade62c897004d07a0d2:action", "state_id": "926db535f0f4d0bc5292da1c05a8633902a27f8c5fe1bd93f300486dd9747e1c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.91796875, 2.27734375, -1.76171875, -1.9375], "student_probs": [0.014383469708263874, 0.9546952247619629, 0.016815979033708572, 0.014105268754065037], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "75aa9906a7dcf6e30f10cb6c7f52036b342bc46f780177d5ca56b7c7af2d302d:action", "state_id": "01416f067d62fe745af779a3177840be29abcbdf07563b7637452e9efa9d0f17", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.81640625, 0.666015625, 1.36328125, 0.66015625], "student_probs": [0.22504466772079468, 0.19362209737300873, 0.38884231448173523, 0.19249090552330017], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bd7c2d3f47654b5e9c6f314fba4a6cd478e18c055d652e4897eefc84a5f2c84e:action", "state_id": "3658474f0db8a027b1aaffe67f46af0aad06e44e1102f764894c731d4d040caf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.6533203125, 1.08203125, -2.24609375, 0.1796875], "student_probs": [0.1089976504445076, 0.618117094039917, 0.022165853530168533, 0.2507193386554718], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2005fa82c8fc70b5851100e58eceb412b746787719d7e3b4696ed88e5568c42c:action", "state_id": "9e381c5b6664df5d298f477f2661aaef00d214e21a18ea1698db7c8bbba97d15", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7109375, 1.36328125, -2.01953125, -0.49365234375], "student_probs": [0.037389520555734634, 0.8088465929031372, 0.0274618212133646, 0.12630197405815125], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "66f158f420b9f016cf92e4c904f2e289ad2ebd33ab8a25e54ee9f471b87f0478:action", "state_id": "a5c83b4b54194a5c613919bc6f2d3157261a2c1839c958f7db15def25d9c3e61", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.99609375, 2.63671875, -2.6015625, -1.29296875], "student_probs": [0.009401264600455761, 0.9664762616157532, 0.0051313843578100204, 0.018991075456142426], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b5e934431098149c41981b9a6da4bd3e5fa4c4ee254dfb61297b15097f6eca7c:action", "state_id": "6b9af706efafc74471c0eb713974c920e009cb9c5b7c11aaed7ef706fd717056", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.984375, 2.90625, -1.95703125, -1.3984375], "student_probs": [0.007306678220629692, 0.9720563292503357, 0.007509226910769939, 0.013127722777426243], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d3c41c2fe934c6f71474e45fc50ccec16aae96ca16188ba5ffc05423efbc5260:action", "state_id": "8662e077ba1137a085ce6b4e20644fe904440a6c7263fabe1f403e7921cfbadd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.73046875, 2.65234375, -2.22265625, -1.2890625], "student_probs": [0.012015032581984997, 0.9619582891464233, 0.00734464218840003, 0.018682081252336502], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "555fcd8e9e6db4c2cc1cb879218f8540516c983166a68232fdee4f5b28b46dc2:action", "state_id": "eb2dfdcf3a7973546ab11be9d0245d96436dd595768f1c4c5acb7a54e8c4f07e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.994140625, 2.75, -1.4453125, -0.669921875], "student_probs": [0.02207871526479721, 0.9333260655403137, 0.014061521738767624, 0.030533751472830772], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c78000c46319b0e8c8970b4b8f307f9599189e68398cd5a928ba2b768029e5a3:action", "state_id": "c8ad172a4ca0b1dcaca043ab8555d172da376e42bea18e5db4a0866fca0590ed", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.47265625, 2.822265625, -1.78125, -0.990234375], "student_probs": [0.013041047379374504, 0.9562541842460632, 0.00957837700843811, 0.021126408129930496], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4b546a42313699e402756de21fa3354db4cc913d346361e5fadf2255f8ed940e:action", "state_id": "6e5b9e705510ad56d75ddda3d5413af93d0813bdf65bdfbcf8b1370554932a9c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.859375, 2.962890625, -1.7890625, -1.24609375], "student_probs": [0.0078024063259363174, 0.9694198369979858, 0.008370759896934032, 0.014406988397240639], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b46894dfe702f1c7bb2f37df0509ab39ea67737e0e4a7be5ca1cb511d5afbd28:action", "state_id": "3096d29bd8d8aff1dda989bedf3272214787eb6325528d0bf03549ccc609c64a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.08984375, 2.921875, -2.05078125, -1.703125], "student_probs": [0.0065072569996118546, 0.9771466851234436, 0.006766476668417454, 0.009579608216881752], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a71861db6789dda7d8e7f064537f061eb881f06b6e4ac83aa38641d8d65b51a0:action", "state_id": "214b1150a629d952bed3702d8f4d59909fe73d2e4b0ee85c36a5bef37735c0d1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.03125, 3.1015625, -2.07421875, -1.6640625], "student_probs": [0.005783865228295326, 0.9803255200386047, 0.005540603771805763, 0.008349984884262085], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dc5b0bf7b667ce6f2199bdba61c5bdb0f37e98d60604040cc109a4db0ef18b96:action", "state_id": "ddba2582c688e5a6c4831ae033158da790289e2b9babaded50d050069d10991e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0546875, 3.201171875, -1.8203125, -1.6796875], "student_probs": [0.0051175691187381744, 0.980967104434967, 0.006469213869422674, 0.0074460189789533615], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c4d8b3155fc22a7af6594f3290d2ccd481de3c75d027342db8a6e505e6cba04a:action", "state_id": "6413a11993531aca822ec907a976e41657d34000573f75b5a41e56aa434bf663", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.84375, 3.208984375, -1.484375, -1.328125], "student_probs": [0.006228325422853231, 0.9744195342063904, 0.008921664208173752, 0.010430483147501945], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4bc1138223e117f7bc0d5bd7319de19771eebc1e9454b5892ad42d0e3ce30a10:action", "state_id": "9cdb786a31a65156d3e6fe22fbc2688a43f6a4591afc68da7215e9212289e8d5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.125, 3.228515625, -1.7109375, -1.61328125], "student_probs": [0.00463970098644495, 0.980600893497467, 0.007019643671810627, 0.0077397446148097515], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "99abea423e707dee693aba7a46fd467d5e304b86ef0d22c739382d60e75080e2:action", "state_id": "f6dd190669b49f11ee1d6b201a1653ab515f88541ac2d4db9e57dba9302e2181", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2265625, 3.166015625, -2.08203125, -1.8046875], "student_probs": [0.004475282970815897, 0.9835296273231506, 0.00517117977142334, 0.0068239918909966946], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b5bd17456aabb2fa5c60dde71dff3389380421af08622a7bb857dc5d0180de4d:action", "state_id": "f7922c29c64184c3eaf2954fa1cbfb775a95226b955ffe78610b8cdb875046fe", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0859375, 3.25, -1.80859375, -1.61328125], "student_probs": [0.004726094659417868, 0.9814555048942566, 0.00623664865270257, 0.007581836543977261], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "884012333a77284aa1420e19b83c9d75e5009597901ed075cfe4c92e305b4ae6:action", "state_id": "cd91a36d3c6625ed75181b5dbb581b46c8cd9267d6a23409adfed5be343f4f40", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.125, 3.23046875, -2.31640625, -1.69921875], "student_probs": [0.004648572765290737, 0.984396755695343, 0.0038387777749449015, 0.007115969900041819], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4be193d8826cbeb9e356c2cb60a75ed9c872e56054fdd78566085b205e21573c:action", "state_id": "b235955dbdde8f78299ff2cbcd30b20ad43b484f859f47de39c0c4573a6446a4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8203125, 3.068359375, -2.15625, -1.33984375], "student_probs": [0.007347074802964926, 0.9755233526229858, 0.0052507175132632256, 0.011878985911607742], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ff76d1dc755b748f149c3b817544434cffdff9fbccea1cdba43eed38df3e0ec5:action", "state_id": "51f7352ef1068464a0fa218bc644b8cc4c1c7c8fa41a6164567d7aa9261bde91", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.04296875, 3.166015625, -2.10546875, -1.54296875], "student_probs": [0.00536203570663929, 0.9807603359222412, 0.005037166643887758, 0.008840502239763737], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "231fcb12611a828ff5e5712c5567f1165a4c25785971f997005deb3d37722460:action", "state_id": "6051eced32ac30eb1b476f29a913d3e323885b39ceb6c636bc2b2fba4f139208", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.171875, 2.939453125, -1.859375, -0.49462890625], "student_probs": [0.015504115261137486, 0.9461807608604431, 0.0077959587797522545, 0.030519064515829086], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c18b205fa41ad47616b03d9e594fd73bc2fe6f9c5668a37985819ed2e3004c9e:action", "state_id": "41cd3c8f939caee67222f13bc466005d0ec915118952dc082a87c6e6eb232c71", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.283203125, 2.837890625, -1.68359375, 0.19140625], "student_probs": [0.03917720168828964, 0.8881927132606506, 0.009657206013798714, 0.06297288835048676], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aa37f23b247a69edac3873d8fc8266a086370254f7e8b4c552e4965ceeda539e:action", "state_id": "0af8451adf4a00827f311649b8309a70ee92413e111e8a859dc95fe676eef605", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.67236328125, -3.453125, -1.359375, -0.71435546875], "student_probs": [0.3962050676345825, 0.024561256170272827, 0.1993217170238495, 0.37991201877593994], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bf049691ed4653921f920ead7781b0fd6ba647df3af7586cbb9142795a0424a7:action", "state_id": "cfdb14c92bb04cc847e0b0a777667ee4dcd7ffd66e0d4c4fc17f81733d70dc25", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.255859375, -2.8359375, -0.57720947265625, -0.4677734375], "student_probs": [0.38314592838287354, 0.02903023175895214, 0.2778456509113312, 0.3099781572818756], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f8f88e88705b33b6131dbd7274ae5eceb049c88cd0bddbc3856eee9e6f6a2180:action", "state_id": "df292ac5d1fc7ab72187af92cc4a5e5f39312b0062e64b5880e7681f01e6fb88", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.6953125, 2.4609375, -1.52734375, -2.03515625], "student_probs": [0.005565972533077002, 0.9657661318778992, 0.01789713278412819, 0.010770684108138084], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7fdb48b6ea264a95ecc0a79af70759f42aab0ffb02d38f0fab5c04b6cca486ca:action", "state_id": "b6937e6006f48907081c5e113e6b8c6b0775cb690826df873dfee150717fd29f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.07421875, 0.50390625, 0.23046875, 0.71484375], "student_probs": [0.1577230989933014, 0.2811718285083771, 0.2139042317867279, 0.34720084071159363], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b3cdeab3df1c75fbc8907fa74536b69668d797ce7d7aca1fc7fccd104cef2293:action", "state_id": "40d4df5dfd563cea7de22292fa22a85031447d272fa0f93e0987cec6e3fc8227", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.947265625, 1.21484375, -0.212890625, 0.19140625], "student_probs": [0.06713101267814636, 0.5833314061164856, 0.13991305232048035, 0.2096245437860489], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "554e00a98123f61f0dcfd1507b89702821f6dae4c7a1d25a9142cf9134c7379a:action", "state_id": "a9df0f3818f63c8c7fbfd9bee9a171517454d7a4a4b437290a290ffe34fbf5da", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.162109375, 1.8671875, -0.2734375, 0.05859375], "student_probs": [0.036358144134283066, 0.7519840598106384, 0.08841928839683533, 0.12323848158121109], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9f119bc7f58a899de0f986ba9af6c1244b469db479ec664bb3420bd1f9687069:action", "state_id": "87f5055c7a5b14bc9c0af2af6671d8afee1ffa4b876a43cc1ab322142b6a7019", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58203125, 3.162109375, -1.19921875, -1.24609375], "student_probs": [0.008419303223490715, 0.9674538969993591, 0.012346092611551285, 0.011780723929405212], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "728e9d8e6cc7a58455e00d43a707496532264e0c4305f23d4bd3add625031ecd:action", "state_id": "3ca33f70fa432dfc19dfdc4bbb89a6e9db1a3b307fe370b7db82c162828ab2a9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.078125, 3.166015625, -1.212890625, -1.126953125], "student_probs": [0.013788900338113308, 0.9610289931297302, 0.012050405144691467, 0.01313178800046444], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bdc6952adba2c9170309941d347ead7c8bbe62c5188ec51e2e6afddffd3e656d:action", "state_id": "307de83f57efd13d7808b4eaae4ba356f2a180554e9ed3107f2cc1c5b93bb36b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.69091796875, 3.384765625, -1.25390625, -0.875], "student_probs": [0.016315318644046783, 0.9608208537101746, 0.009291649796068668, 0.013572183437645435], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7a5684164ce4a83b0f570bf63c700e371c75da7e7c3c4c06cfd08e21ac91477f:action", "state_id": "460e944201f32324682afb64786f0b5de965bda31d1514c1d3d24edbd36796ff", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.741455078125, 3.416015625, -1.23046875, -0.947265625], "student_probs": [0.01507456786930561, 0.9634107351303101, 0.009244191460311413, 0.012270507402718067], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0ea715c33a11504facbcf31c91609dc0ec986f89ea4a149d3c9670d6813bea60:action", "state_id": "911deda7913afcefac578df9839bce90be5189ef0c1abed05c5e67dc3c0c74a6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0859375, 3.03125, -2.26171875, -1.78125], "student_probs": [0.005880262702703476, 0.9812124371528625, 0.004932372830808163, 0.007974819280207157], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f05ce608a78ebbeda45568006eff2ef7dcb8922fa82234545f5af024ac0b3065:action", "state_id": "82830c00d94be5d5523a2c666ba07b9b31a4b4cc096ff68d60b66d5a36db3478", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3359375, 2.908203125, -1.6875, -1.083984375], "student_probs": [0.013757814653217793, 0.9588624238967896, 0.009679831564426422, 0.017699919641017914], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3604a690146eb95c4c1a22361052f2c14f918021bbe8859db0db46c47a1d0eaa:action", "state_id": "fc3b6371950b67314d573128442ee0535ab4e9df9100bcbd5d9251e25f71e67c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.484375, 2.94140625, -2.05078125, -1.21875], "student_probs": [0.011567395180463791, 0.9667806625366211, 0.006565207615494728, 0.01508672721683979], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2ba12f6a4692ee04627e65ad42fcd068c981692690b805f3efba1479ddfd87f5:action", "state_id": "045aa6d5e4267f7a202e184581236e214b2da97191cf31707cc02aa5cabcdc83", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.0078125, 2.759765625, -1.8515625, -1.16796875], "student_probs": [0.008188726380467415, 0.9632725715637207, 0.009573590010404587, 0.018965130671858788], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bb79ebf74bc805752a2318ebf0fe6527651963fbd2d145760b29ac27bfebd661:action", "state_id": "bda074e787cb5bc8f86d6860d73ec1a9193e119cc6885fa08be0a0db61bf62e2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.15234375, 2.953125, -1.6796875, -1.36328125], "student_probs": [0.005891817156225443, 0.9716864228248596, 0.00945194624364376, 0.012969842180609703], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f85b9798476729149557ea1125b01076e9f3ab37812554ad32c847fe354baa91:action", "state_id": "24d0822d5f9f921410d229c6bbee19bb3ab702ccb892f6bfd72c75d5606eafc2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.2890625, 3.087890625, -1.78515625, -1.66796875], "student_probs": [0.004527382552623749, 0.9795536994934082, 0.007493606302887201, 0.008425287902355194], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "88fd551a2012ea6e072b5e805c91f2ec857d5e7ef6d0d4dd1682cb432698bb8b:action", "state_id": "39fbd4e089c44b8aeeaf652e308846e14278888fa1416c55b51a02a3b6265043", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.47265625, 3.1796875, -1.890625, -1.8515625], "student_probs": [0.003452929202467203, 0.9839417338371277, 0.00617960374802351, 0.006425771396607161], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "af3198153a8dd2459cc55b9879545b6003ccdcacbab2ff0be5ffd574a5688d9f:action", "state_id": "20e23f834454121bfa052f014b84c96c633405985a46faa45873a53a724e49ef", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.484375, 3.1328125, -2.265625, -1.8984375], "student_probs": [0.0035822303034365177, 0.9855235815048218, 0.004458157811313868, 0.0064361016266047955], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "59d6c2fc1eb34dd0475d50c62b9518d5f0ab213ca00d0bfa4c62679c7ae4541e:action", "state_id": "109ad661e6f80f55d22677ae6181e28c952c7f6ae83c60c3b0b6980ec38b01f1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.453125, 3.203125, -2.05078125, -1.78515625], "student_probs": [0.0034421104937791824, 0.9846978187561035, 0.005147074814885855, 0.0067130508832633495], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "497c7b81b84b637faa79e8b308e3684b337c9a6d078d93398270508a595faebd:action", "state_id": "b2b7172b6989a738bbad03e0c16b5994ef865af3dd919ce4aa4c15a9a53b462e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.515625, 3.19921875, -2.08203125, -1.890625], "student_probs": [0.003249413799494505, 0.985666811466217, 0.005013169255107641, 0.006070704665035009], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7f0bc64bde1fecdfd4673c12fa371ce3acb6d60d5f2b73ed7dfeef259825bdc7:action", "state_id": "462213461d2038bf85f28da76c70a314c60d695ed307d16fe4540a22ba4fd36c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.51953125, 3.21484375, -2.1328125, -1.9296875], "student_probs": [0.0031888217199593782, 0.9863649606704712, 0.004694399423897266, 0.005751698277890682], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f15f9dc8b82f5d84427421e59458ed30aa4bcd89e396ab5f526c1ab8f624fe93:action", "state_id": "544f2208b4fc9f5d084759e7aae8ed7c9751bd8b86e43ae1aadbcc2ce346aa91", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9296875, 3.15234375, -1.8359375, -1.640625], "student_probs": [0.006077755708247423, 0.9791322946548462, 0.006675108801573515, 0.008114868775010109], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "373374bfcc4df16fa4549eac31119b672de155cea3adbb8fcd961662a17fa780:action", "state_id": "bccb9d58f953f546c2d00d770458f6993616e697e58f49b4174c91650f7d548c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.13671875, 3.16796875, -1.875, -1.7578125], "student_probs": [0.004877145867794752, 0.9816625714302063, 0.006336198188364506, 0.007123978808522224], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "772c55bc43e080ecbc7bd56041df5a602a3dac7f6e6c29151b4d12dc1dfb5e29:action", "state_id": "9ba07f82f77f552a7d686a2ce5745562baaa781eadbc0ad87d2b155a980aa722", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.11328125, 3.1875, -1.5859375, -1.5390625], "student_probs": [0.004878916312009096, 0.9781904220581055, 0.008266960270702839, 0.008663699962198734], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e7af888241a93292adbee3f1f580f7abb8d0fbecb2fe8f6f167f68b6ca2e410a:action", "state_id": "c2b73192b94602c9e97b93e3bbd78ddb89b25a43e20ec572f07f7f8e4fd44d55", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.9453125, 3.265625, -1.3125, -1.2890625], "student_probs": [0.005316992290318012, 0.974422812461853, 0.010011358186602592, 0.010248771868646145], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "71e7de38432cbac3a2f94e1c9085383e962f8ae680d77a75ce2bdb6bd56f3756:action", "state_id": "607711eda14ed621af52fd2d64d0b9584d38ccec598aef0fdf6fda51934ef98e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.8515625, 3.1953125, -1.171875, -1.1328125], "student_probs": [0.006228170357644558, 0.9687026143074036, 0.012289806269109249, 0.012779376469552517], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d8ad186c9d76034a44261fc831a322a15ff3981b782b1fa8ffb291a0330234e5:action", "state_id": "12ea28ed28f1fa2e6959126f083911fb5ca035fa034b95a5025c02fe3d5c1829", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.51953125, 2.927734375, -1.78125, -0.9921875], "student_probs": [0.011254003271460533, 0.9610145092010498, 0.008662515319883823, 0.019069069996476173], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "77ce51cc9d798ba22bf1c7cc08a87d8ca35df736634aea0532e57ce4c549c72b:action", "state_id": "a5c9ac734c53253267dd183492cb39aed2c168cbdbda627306bf187705ed26ae", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.56005859375, 2.462890625, -1.65625, 0.05859375], "student_probs": [0.042118776589632034, 0.8656172156333923, 0.014073620550334454, 0.07819032669067383], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a4a05a411146847304eae412477be8aca06d358345f768dda50174dc443c15ef:action", "state_id": "1c1c22455538a05889807934725b6126399b4ab71e232958fd085c1681b24e6c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.08203125, 2.392578125, -1.52734375, 0.34375], "student_probs": [0.07949689030647278, 0.8013235330581665, 0.015900379046797752, 0.10327926278114319], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "67822a2751f8ff4a01e787a322b388da0709d93f17184a57d6414a53bd126f04:action", "state_id": "a2a572558396dbff1c28185dfe447299d4fe1a20854b14fef73b334221e1d09e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.36328125, 2.623046875, -1.74609375, 0.28125], "student_probs": [0.043537385761737823, 0.8625975251197815, 0.010922310873866081, 0.08294281363487244], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "87db5c1d52a008d86980f7b7bbf327f2b12d914b30f223baa42475e61d77f8ee:action", "state_id": "7cc0ed3cbae7418e6fb2dee149ff770cc1f792e1335563b690fa994ed72ed4b3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.91796875, 2.611328125, -0.9619140625, 1.046875], "student_probs": [0.1294011026620865, 0.7036466598510742, 0.019747642800211906, 0.14720456302165985], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "277b6e301c7562e5e0ad548b8ce8b4600407d5e5e8400a19599aa70e30b79d67:action", "state_id": "c82be5cbb50b474b56ca509f537653163a87cfcde8a29c2ece7a30dc88b8d529", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.58984375, 2.521484375, -0.7384033203125, 0.98046875], "student_probs": [0.10369498282670975, 0.7155807018280029, 0.027473080903291702, 0.1532512605190277], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cd182f7d573b4c86c1e734a1e30a09956c453148315fa0a6d8eb649023f03f3e:action", "state_id": "c271e131243c4c0ba8788529dee7bcee92f07d87ce7b09d4a5f974bd797669d5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.49609375, 2.884765625, -0.63671875, 1.009765625], "student_probs": [0.07198100537061691, 0.7845216989517212, 0.023186955600976944, 0.12031030654907227], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f900df0be8f4e3e7877d543299097faa908528b6c651047dd1a40cddc9d3749f:action", "state_id": "ae088c500b19940c70fc0fa677318c59863dd0eb4f1f595834419b87d22da2ed", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.1796875, 2.91796875, -1.29296875, 0.4140625], "student_probs": [0.03954877704381943, 0.8758466839790344, 0.01299095805734396, 0.07161358743906021], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d5157efae8d396bf716b701ded7eaccd442f99fce5fa9a4a70a22bec2bc3bea9:action", "state_id": "03869c77f2db001075ccdd2198f0094bb969b575d4fec1fa71fdf6f598af55ac", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.1875, 2.22265625, -0.9365234375, 0.724609375], "student_probs": [0.09354999661445618, 0.7159799933433533, 0.03040090948343277, 0.1600690633058548], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "df5cc3d38b9de84bcf5bc8ec5f85f50f718d718d50762a3e4fc0da2cd0c701a7:action", "state_id": "44bcf34c3ae3401538cd76ff76822d9b21582055b36351d832a9ee0eec582327", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.078125, 2.474609375, -1.052734375, 0.62109375], "student_probs": [0.06160787492990494, 0.7911788821220398, 0.023247098550200462, 0.12396614253520966], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b9b557b9901e747c8dd113df70fa460c927d268cac35bced6d3688e1ced86283:action", "state_id": "c67727655845cb085282d718f3a414b032a6758833b461a413dba513818ffe3a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.4228515625, 2.544921875, -1.45703125, 0.46875], "student_probs": [0.0430234856903553, 0.8367452025413513, 0.015295619145035744, 0.10493569821119308], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a4cec942f92faf8f9bad0b4c35683f00907355932a70fb84392a5227cc57a317:action", "state_id": "615f062ff99aa2ea3dd63fb6f02da43a6d6190010241e8841dd7e4b7fb710e26", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.63525390625, -3.46875, -1.43359375, -0.6217041015625], "student_probs": [0.39642828702926636, 0.02331271767616272, 0.17842267453670502, 0.4018363654613495], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b45405a38d36b7c505fdec377ca8f6c35567bd80618ac2ef03b64f960e994865:action", "state_id": "3302177b2ca0394502d5c08d0ea47a894c955d55cdde1e24f1a113c95f06903a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.16015625, -3.125, -1.33984375, -1.19140625], "student_probs": [0.339565247297287, 0.04759950563311577, 0.28371739387512207, 0.32911792397499084], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6b7f9a1195c0827da4cb0aa00b12f542fc136ee0e415d64acbdebec0e567ecb8:action", "state_id": "35dacdbf1376d1bf855c5b2e7a3914558b7da7377dce1fad30d7a19e53c9c51a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.984375, 2.5546875, -0.528076171875, -0.9453125], "student_probs": [0.009830945171415806, 0.9202060103416443, 0.04217526689171791, 0.027787813916802406], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "718604e955caffd31ba8c30101ef9132c2d15c1e5f3f391c352a557456f91f34:action", "state_id": "d32142df993378b0c51a977d627dbfa40aa53cf435d651a8994fade45373448e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.064453125, 0.38671875, -0.193359375, 0.25], "student_probs": [0.087870754301548, 0.3750423491001129, 0.2099691927433014, 0.3271177411079407], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c6cc3a5218dea31bf3333346fdda79c35d14c46b3264d2eedd2ba7fc3bac8f42:action", "state_id": "bfce3eaffb800ce9f0acc89c8dcd8489f737f934d0672ac0fd783b2c209045f3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.009765625, 1.193359375, -0.015625, 0.4453125], "student_probs": [0.058683790266513824, 0.5312796831130981, 0.15858714282512665, 0.2514493763446808], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "75be1dfe4243d8fa67269ef29e229cbcd9dd89ab78bf2ad5a84fb3f7d698b7a1:action", "state_id": "39fe4603b7343a4a529889dc3fb789c32723de0ae9a381569e462e3bb99b8b8c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-2.140625, -3.8046875, -0.98046875, -0.9453125], "student_probs": [0.1301339566707611, 0.024643220007419586, 0.41518348455429077, 0.4300393760204315], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aaa6ca92484bdf51c0f05e4335d2400616467ead6cc72f1d5b4fe0dd0c85c20e:action", "state_id": "c8a1360457acb848e4a18bd6ad69f1ecae2897834b8f6139303f2a02ab2943c3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.859375, -3.359375, -0.931640625, -0.92578125], "student_probs": [0.15884266793727875, 0.03544259071350098, 0.40167713165283203, 0.40403762459754944], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"}