{"id": "7675c640eaa1bcdf20d122b9e9773d80a7d98009767300ff4e305d8d5e59e50b:action", "state_id": "c15fc2e39ce68a7ff7763e5009b0e913cbfb834b0f62246d626226e5756585b3", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.04, 0.22, 0.67], "teacher_probs": [0.07, 0.04, 0.22, 0.67], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.705322265625, -0.408203125, 1.03125, 1.08203125], "student_probs": [0.07144160568714142, 0.09615866094827652, 0.4056345522403717, 0.42676517367362976], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.04, 0.22, 0.67], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "73fcb745f0156d36f95be4b6e038d3e951ba24d0b38f2c5e15653d835fdffdc6:action", "state_id": "3fc193943a0b05556150ad20505d8f76e00acb0e8d6d18662f171be6b8652f6d", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.12, 0.52, 0.36], "teacher_probs": [0.12, 0.52, 0.36], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.66162109375, 0.458984375, 0.1875], "student_probs": [0.15614505112171173, 0.47885167598724365, 0.3650033473968506], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.52, 0.36], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "90a50a53be7feb063fb97a89cdd173e2b75c669ac0b76c72a81aadd271b0152d:action", "state_id": "5e40bea098f3e1478460956193a896efabd390db06b0f19f172a60a6c7e5fc8e", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.19, 0.36, 0.45], "teacher_probs": [0.19, 0.36, 0.45], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.455078125, 0.65234375, 0.38671875], "student_probs": [0.1575527936220169, 0.47684067487716675, 0.365606427192688], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.19, 0.36, 0.45], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "307baa076ef97e63a0ff39ef3888c8d49b4e9f63cb41a96d29128b24882b5436:action", "state_id": "5bc7c0c8cbcb357b80e29334c23774d63482e65d9cd17a055796d007576922fb", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.16, 0.6, 0.24], "teacher_probs": [0.16, 0.6, 0.24], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.015625, 0.18359375, -0.8466796875], "student_probs": [0.18176597356796265, 0.6030129790306091, 0.21522095799446106], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.16, 0.6, 0.24], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "af4816b73d42669eb8c3d38c4f0954206a9ee8c9de88de4c4b36b16f16ed2400:action", "state_id": "70e64bf5578bc46280f6c9901707bfe3764a4df553b81ce9162e63fc1a52a8e7", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.31, 0.69], "teacher_probs": [0.31, 0.69], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.978515625, -0.8984375], "student_probs": [0.4799911677837372, 0.5200088620185852], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.31, 0.69], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "57dbe1c126e89a4513d4a55acf2347d6d7b7b3c786fb075f848da88da60ecacc:action", "state_id": "1ac4254cc6e960bf7fcd724d618cb5ee1d1b862e926e98ab63a168013c41eae7", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.12, 0.76, 0.12], "teacher_probs": [0.12, 0.76, 0.12], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.7490234375, 0.08984375, -1.158203125], "student_probs": [0.251386433839798, 0.5816439986228943, 0.16696958243846893], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.76, 0.12], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ebcbf0a7ffd650474acd579bbcd07004282bd36681d1832889b84241efd3d3f3:action", "state_id": "40962fa89d0f85937db1ce837d7ce215e84d3afffc1dd07d8029e92d6d6740ce", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.58, 0.36, 0.06], "teacher_probs": [0.58, 0.36, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.580078125, 0.04296875, -1.107421875], "student_probs": [0.5651580691337585, 0.3302982747554779, 0.10454373061656952], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.58, 0.36, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "73a9b7f9ddfe2e2b0e73855539e7db08530795ad0b3a23e6c3b245a32effd0bd:action", "state_id": "eeb0b8706f1047c4c36219603dbbdc15814e5e010757937083f71a513ec5772c", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.71, 0.29], "teacher_probs": [0.71, 0.29], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.05078125, -1.10546875], "student_probs": [0.7606506943702698, 0.2393493503332138], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.71, 0.29], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6ea8b278085750c4f1d74afa05225e804e81ac67b5b610da6b0c89adc9ed2a9a:action", "state_id": "51cf0e917d65c40360775bfc002e5c47e4cdb0aaf152ebcd508aaa3326343542", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.26, 0.67, 0.07], "teacher_probs": [0.26, 0.67, 0.07], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.19140625, 0.177734375, -0.8505859375], "student_probs": [0.4275016188621521, 0.42169663310050964, 0.15080171823501587], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.26, 0.67, 0.07], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fa2ed11507b972fb85f1cbdfe2e92d2e13a7c75b2697b430affead3830618630:action", "state_id": "700cf1d83329f2157208e468d9b74b920d9044b19550c6204eee2bc67e855b02", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.82, 0.11, 0.07], "teacher_probs": [0.82, 0.11, 0.07], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.40234375, -0.896484375, -1.03125], "student_probs": [0.6616812348365784, 0.18054062128067017, 0.15777818858623505], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.82, 0.11, 0.07], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "690562a1b80308397bf923f308b0143d5ae465607ebd07e6165f183ad87b4cb4:action", "state_id": "9b5d920d3b21bf5d1c1d3746508966abc10c67709ce4457ca2a844b696781958", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.39, 0.61], "teacher_probs": [0.39, 0.61], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.958984375, -0.51953125], "student_probs": [0.39187130331993103, 0.6081287264823914], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.39, 0.61], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "21a826a9424e19820846e652eb2e0290af6bbcd8b58a166c76a834ccfd5cf389:action", "state_id": "a85616fc5e4f386e2aaa20fc9c3ae66d294bb201f5023ea048e47a51fc65fd42", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.91, 0.05, 0.04], "teacher_probs": [0.91, 0.05, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.689453125, -1.15234375, -1.3046875], "student_probs": [0.7724018692970276, 0.12245064973831177, 0.10514751821756363], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.91, 0.05, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4a5e050d3f3dea3f9e092d21dc4bb268993021a1d46d3ace70eed4b31656f5cd:action", "state_id": "c915bf47daf56c8135c7c3dd6f89ced9193a54b27f4d12f6aa00f7bfc4f7b540", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.57, 0.39, 0.04], "teacher_probs": [0.57, 0.39, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.08203125, 0.455078125, -1.05859375], "student_probs": [0.6054007411003113, 0.3234153687953949, 0.07118382304906845], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.57, 0.39, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5a58e2ec66e5beb141113892b5449575dad5e603ddeb15f44713ae1feb1fa4db:action", "state_id": "641b85da39ca96109d6ac7fe0cddd877d7c8734e23d6e8c97f8c814047240147", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.77, 0.23], "teacher_probs": [0.77, 0.23], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.158203125, -1.033203125], "student_probs": [0.8994751572608948, 0.10052487254142761], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.77, 0.23], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e8834db780a0bb82d30f9f13d80e8c3e660a4dcd2cca5e4c5c8f36a3c413f8c7:action", "state_id": "a20e107eca453c677355f966cffedbf1cc4919f60672b1bff35f424ef2f33735", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.62, 0.35, 0.03], "teacher_probs": [0.62, 0.35, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.083984375, 0.796875, -0.91796875], "student_probs": [0.5303630232810974, 0.3980001211166382, 0.07163678109645844], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.62, 0.35, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b8ec9f568252f79db15034e898a162221f6774277fb965c23f75f7fbfcc8ff97:action", "state_id": "2bb7744ea729f91b659a88d3e60f58df081b25c1aa0b9367345879cf26f13a48", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.72, 0.28], "teacher_probs": [0.72, 0.28], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.748046875, -0.943359375], "student_probs": [0.8444089889526367, 0.1555909812450409], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.72, 0.28], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "16e45c4c159ff4036e7d7e6d3cdf8f4d402144ccf7eb10113693d07a30f8d94d:action", "state_id": "d84439561547b7a869523a00d3bfc89dbdec5b8c2ef89757042bd496b592297a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.35, 0.27, 0.38], "teacher_probs": [0.35, 0.27, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.576171875, 0.47265625, 0.44921875], "student_probs": [0.35939720273017883, 0.3240547776222229, 0.31654804944992065], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.35, 0.27, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "81fb4fef5f5221ea14eabace10b23107270b4d6b554fe8a8cfa8816e4c944b85:action", "state_id": "9d7e5ce8487176a7cca0a68764b79daf8255f60f88b02f75db152812364e69e2", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.59, 0.29, 0.12], "teacher_probs": [0.59, 0.29, 0.12], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.328125, 0.46484375, 0.12109375], "student_probs": [0.3378949463367462, 0.38739845156669617, 0.27470663189888], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.59, 0.29, 0.12], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f629f86a9579bc87f3f284e4be92b50b55b8f26216ea09ede2707bb9c47d7c79:action", "state_id": "8b09189aa5b659b82471e62f33667196cf05ca7cfed8b11d30d5876c3f59eb39", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.69, 0.31], "teacher_probs": [0.69, 0.31], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.453125, 0.4296875], "student_probs": [0.5058591365814209, 0.4941409230232239], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.69, 0.31], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b829a0684053a68d0cab380077c5cb5eb897bced937ec03106a62d06e3faba36:action", "state_id": "91ac6e37df91b673936e024070d25cdfc5f60d3fd5f4170e11bce6edba87cc23", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.29, 0.31, 0.4], "teacher_probs": [0.29, 0.31, 0.4], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.650390625, 0.41796875, 0.830078125], "student_probs": [0.33450913429260254, 0.26513582468032837, 0.4003550112247467], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.29, 0.31, 0.4], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "51f54f09a595869881e531585131eb1c0a685c8c0b937ed2bb46d04b4f365e9d:action", "state_id": "0a4aa10fe970e3f3684bf78c35dae90e348b07e8f1c324aac270a163fcc53c2a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.41, 0.59], "teacher_probs": [0.41, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.67578125, 0.54296875], "student_probs": [0.5331544280052185, 0.4668456017971039], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3175ef5ff7e0d14090ccd944c69ebc8401628d0a3250158f4106abffdc380c28:action", "state_id": "7d03a9c98af5becb8daf0fa7f06693ad9ae13f71cb7725250eae32fb37bbd5d8", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.15, 0.3, 0.55], "teacher_probs": [0.15, 0.3, 0.55], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.05859375, 0.521484375, 1.107421875], "student_probs": [0.16679568588733673, 0.29792678356170654, 0.5352774858474731], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.15, 0.3, 0.55], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2f7002fef8d074b47726dc6de80594473716157509a687789486a48f68141cd1:action", "state_id": "a9de808bb87958e89e8b84090b745fe62c24e60217b897bae7c45fa0c8ec0ec6", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.25, 0.44, 0.25, 0.06], "teacher_probs": [0.25, 0.44, 0.25, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.12109375, 0.62109375, 0.24609375, -0.3359375], "student_probs": [0.2264990210533142, 0.3734337389469147, 0.2566570043563843, 0.14341023564338684], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.25, 0.44, 0.25, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "65e82ea8cc79762609858fccb5f902e86bd8e8199dff7b343bc8ce8c31eac755:action", "state_id": "30efaa40a90a662d77eca9125d13ade0cf42f2fe8389d193aa7a58550642ac51", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.49, 0.42, 0.09], "teacher_probs": [0.49, 0.42, 0.09], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.736328125, 0.294921875, -0.787109375], "student_probs": [0.537318766117096, 0.3455665707588196, 0.11711473017930984], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.42, 0.09], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9e98c46927230d8672d21c34be0aa812cc9e39e82d49590d0612497adf5bd86a:action", "state_id": "942b4f129a90304305cfa2c4498a237e0ce8dc5c0675c6328458633d5fcf5aa8", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.6, 0.4], "teacher_probs": [0.6, 0.4], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.09765625, -1.03125], "student_probs": [0.7178038358688354, 0.28219619393348694], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.6, 0.4], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1e92f398184afe69d9bee00a32ce4cd130f740b0d53938c7e5258e319ef7a2b5:action", "state_id": "c739d02cdf1514de417bca15b20da7b94caf6cc5227b2367b83fa8cd8a221f7e", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.66, 0.25, 0.09], "teacher_probs": [0.66, 0.25, 0.09], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.638671875, -1.00390625, -1.44921875], "student_probs": [0.7590542435646057, 0.14686225354671478, 0.09408349543809891], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.66, 0.25, 0.09], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "10a9f737a8bf33986cf164ba5aee4340044b1c9b2f25768a892e7f2041b14f27:action", "state_id": "b6f9aaff372bea3d3adbdc26d012005faa13fa4af2552ba08dfc15ea5b824832", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.58, 0.09, 0.33], "teacher_probs": [0.58, 0.09, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.39453125, -0.96484375, 0.37890625], "student_probs": [0.44616609811782837, 0.11458493024110794, 0.4392489194869995], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.58, 0.09, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f7d80c3d550abe50ce4f87e01137400a4d59f7cedbcd808b29127fad0dd78e59:action", "state_id": "9e26123c4252c55e0252e76ce107d420cd1a136e1b4dd0ce20ef28519f4f1bec", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.39, 0.11, 0.5], "teacher_probs": [0.39, 0.11, 0.5], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.23828125, -0.916015625, 0.513671875], "student_probs": [0.3798924386501312, 0.11977215856313705, 0.5003354549407959], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.39, 0.11, 0.5], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "600da8a75a8bc4aa01bec814e1d7eeca558e1435043bf5acb5e25be6e690c29a:action", "state_id": "74f5d9ceda23d7e44a91b9333c7118735d849949120953708eb8f3b7d071e873", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.84, 0.16], "teacher_probs": [0.84, 0.16], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.21875, -0.796875], "student_probs": [0.7341195344924927, 0.2658804655075073], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.84, 0.16], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "187300f2674ddb59e04726d4dbffbdae184d7479838134806d07ea4a40150604:action", "state_id": "c5d423dbad3100b8c75a8ad1eae9a95092cadc28c26ea64ae222987b6fef1403", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.25, 0.09, 0.66], "teacher_probs": [0.25, 0.09, 0.66], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.6826171875, -1.021484375, 0.45703125], "student_probs": [0.20668646693229675, 0.1472800374031067, 0.646033525466919], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.25, 0.09, 0.66], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2f920b07d117b57e05f69d91414dafb7d074976aab506e81dae04610cff1b109:action", "state_id": "63d12064ddf30a6b547b31d95dd5317c3f36ffc3fd4c64899027168451a56aa9", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.66, 0.34], "teacher_probs": [0.66, 0.34], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.626190185546875, -0.8583984375], "student_probs": [0.557792603969574, 0.44220736622810364], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.66, 0.34], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4ec7dcbc005ee27d4f2c12926ba525b3a2b931b88dd2a74ceb228404d0d1e6ea:action", "state_id": "3678fa65c5362ded581bd35c40d6ad9bc22c52820be7e5f79397926afec96fbe", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.1, 0.12, 0.78], "teacher_probs": [0.1, 0.12, 0.78], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.08984375, -0.83984375, 0.27734375], "student_probs": [0.16107408702373505, 0.2068232148885727, 0.6321027278900146], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.1, 0.12, 0.78], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "30f59caa3a4125373593df5c958339771b0f366ddb780f287f9ff3db68be9282:action", "state_id": "c2f3b53ec108c173fae4440a63d38004036098a5d5fd1bdf07cbd1be12aafa9c", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.08, 0.34, 0.58], "teacher_probs": [0.08, 0.34, 0.58], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.75732421875, -0.029296875, 0.830078125], "student_probs": [0.1255962997674942, 0.2601087987422943, 0.6142948865890503], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.08, 0.34, 0.58], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e54bbf394cdcc1f3ed75ab5b2f4c22de9b8ba9ffa03e9a5d3f2e4069dc79a1b6:action", "state_id": "aeee7b8165c96e007266e557637e495e028c86e65164d91e088605ed8e6f127f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.26, 0.74], "teacher_probs": [0.26, 0.74], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.7958984375, -0.1484375], "student_probs": [0.34356194734573364, 0.6564381122589111], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.26, 0.74], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b326789ca762d0134a485c00d08b9d5eb6acc26dd4084d873def1fd89b429445:action", "state_id": "4e2a208310da887d50b0da4798b59cbe732e42b4a1f828fc56dd756f0a939b77", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.38, 0.58], "teacher_probs": [0.04, 0.38, 0.58], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8291015625, 0.21875, 0.552734375], "student_probs": [0.12765319645404816, 0.3640054166316986, 0.508341372013092], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.38, 0.58], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "09c705e4087cc4cbf6c88e9020492901c70ee95cbab78201ff2c3515b23ff85a:action", "state_id": "598bc0b3275c222804c8016144f2aac1ddd250b34a1a5792a94726b707d67691", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.18, 0.82], "teacher_probs": [0.18, 0.82], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.78076171875, 0.240234375], "student_probs": [0.2648334205150604, 0.735166609287262], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.18, 0.82], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "158349161fea47ba388226326b7521801057b6efb7b64e3a3993faa2a933867f:action", "state_id": "a282817e7cb0a1505a36a7d997418bedfd195798a5661cb63f083254b04d78bf", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.09, 0.84], "teacher_probs": [0.07, 0.09, 0.84], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.501953125, -0.57373046875, 0.984375], "student_probs": [0.15744134783744812, 0.1465366631746292, 0.6960219144821167], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.09, 0.84], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "171bff4d199f5e55ea2873616194c886d3f7f6453c956ed36c90db5225dfd9b2:action", "state_id": "3a7d5b452fc2985144eccfeaf515f0142634d8d2891249f762040b14161af0ed", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.49, 0.51], "teacher_probs": [0.49, 0.51], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4951171875, -0.628173828125], "student_probs": [0.5332151651382446, 0.46678483486175537], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.51], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a469341d0d13533face8800276403f5802496d090d27ce8dc74cb42c05f0275d:action", "state_id": "06705db861a4d3002f6d2c6e64e32c26d22bd733fd7963d115ef9514cc3680e1", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.06, 0.87], "teacher_probs": [0.07, 0.06, 0.87], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.125, -1.41796875, 0.98828125], "student_probs": [0.09978649765253067, 0.07444527000188828, 0.825768232345581], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.06, 0.87], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "44883fbf4e3477b07c22a89fc65052a1aed78c7302b693322208fd612f756a6f:action", "state_id": "e5de40ad001cbf79012deaac0ebb16556d87301d1f3a096ceb32318c8e5ff6be", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.39, 0.02, 0.59], "teacher_probs": [0.39, 0.02, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.552734375, -1.44921875, 0.421875], "student_probs": [0.4969159960746765, 0.0671190470457077, 0.4359648525714874], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.39, 0.02, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d4d03a772c343810d9690a20cdfe029d2df7ac0baf166423daff75361b9ffffa:action", "state_id": "9bd7e683e0131bbbf2a9abb523c4171525a9ee4960a6050db977ef82a76ade11", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.41, 0.03, 0.56], "teacher_probs": [0.41, 0.03, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.638671875, -1.3046875, 0.681640625], "student_probs": [0.45722076296806335, 0.06548407673835754, 0.4772951900959015], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.03, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6edb57c0b0e850395c6819dcae1f0b4bcd681b59a7dd634023a3ac2d6625d738:action", "state_id": "444563468293151c9f54de8a92e4c08e62865caa73ece5f5dfc2285d4175c649", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.1, 0.01, 0.89], "teacher_probs": [0.1, 0.01, 0.89], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.4609375, -1.21875, 2.939453125], "student_probs": [0.07627774029970169, 0.014220628887414932, 0.9095016121864319], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.1, 0.01, 0.89], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6b9b8996b388cb5abfeafe0376f4f286dffe91bddfa0721c71deaa017e79d0fb:action", "state_id": "e3ad23dcb7e95aff35c8d789dc21f6e0acc6acd302a04fbf5db1b7a245387906", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.96], "teacher_probs": [0.04, 0.96], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.125, 2.8515625], "student_probs": [0.018404891714453697, 0.9815951585769653], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.96], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "74e5ec2fa29a4dfa7fff934fd96fc6e52a61f3ea8d67ff6d1e40a0c94d04a100:action", "state_id": "f42ac6bc1b291397b4479b6ff086056556652fdf819dc285a8bce53be2f406cb", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.37, 0.51, 0.08, 0.04], "teacher_probs": [0.37, 0.51, 0.08, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.177734375, 0.609375, -1.138671875, -0.48681640625], "student_probs": [0.3009887635707855, 0.4634569585323334, 0.08069419115781784, 0.15486009418964386], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.37, 0.51, 0.08, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c1ad8dac1c22023ab9536b7ed92ca6a7da37f6a598b5b5f9699a19dd0988033a:action", "state_id": "b183c3c1e7e8d8e777a0637a5775f78fec20a0e4c67f16bf210b544f0b1e4d84", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.65, 0.29, 0.06], "teacher_probs": [0.65, 0.29, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.4296875, 0.4375, -0.0546875], "student_probs": [0.38110843300819397, 0.38409748673439026, 0.23479408025741577], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.65, 0.29, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e32c7db2382f4a08028c96df6f58a41f4c2c4424c6f14001e61a007594726dae:action", "state_id": "7c377e5b5b534c4294cd59cbe702ebb215e41c2b6ac222e9561d128fc5965695", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.74, 0.26], "teacher_probs": [0.74, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.0859375, -0.400390625], "student_probs": [0.5779718160629272, 0.42202815413475037], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.74, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "75d37225c52d32164ba97767515d0d3969e401c821c7be0ace2b11a11ddf4383:action", "state_id": "4df90b1b7dc9bb144434501a6229fedbac1f7901ffbcb96c68a487f29bc313e9", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.49, 0.45, 0.06], "teacher_probs": [0.49, 0.45, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.279296875, 0.314453125, -0.72607421875], "student_probs": [0.4163734018802643, 0.43127188086509705, 0.15235470235347748], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.45, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "99364f5bdc714bf8223619ac257d302f50f344b8b358a06196d027dbcc97b266:action", "state_id": "48ecdad2adc2734296a0d1655992596980dad58868292ab00ac4027462740e94", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.3, 0.29, 0.41], "teacher_probs": [0.3, 0.29, 0.41], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.671875, 0.16015625, 0.26171875], "student_probs": [0.4418891370296478, 0.26489678025245667, 0.29321402311325073], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.3, 0.29, 0.41], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e8109de07b5160a2c1d6d5110f5e5a04fa2dd4cd43a5d65ea2e6cadce921a90e:action", "state_id": "c412a5e85deb087a81950bc65ff63b43ac512921fb19833e7fcbcb3118d60ee9", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.5, 0.5], "teacher_probs": [0.5, 0.5], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.76171875, 0.23046875], "student_probs": [0.6297746300697327, 0.3702253997325897], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.5, 0.5], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9b12025c2d3f07d34163631a6717e3e1babd11d0f6c199da915b48a5ef0de95d:action", "state_id": "de07e7e669de36c0aa0b04d7754954db2c643a5fb5fb5664d37f8968662a84f6", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.24, 0.34, 0.42], "teacher_probs": [0.24, 0.34, 0.42], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.32421875, 0.763671875, 0.783203125], "student_probs": [0.2418774962425232, 0.37535959482192993, 0.3827629089355469], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.24, 0.34, 0.42], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ee400df386717caa084a7974cc8e318180bf3624f584e0fc1d9634252b883639:action", "state_id": "e62ddacbc9aa1f14f998f4efb48eafe6d1964781365c10302f508fa43241edb4", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.33, 0.67], "teacher_probs": [0.33, 0.67], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.3359375, 0.8671875], "student_probs": [0.3702253997325897, 0.6297746300697327], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.33, 0.67], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "700308cdef1970233d77da58ec74d1aed49579c76be13bb3ef5c0e875bd84531:action", "state_id": "bcabfa55a100bd5994fa84832fdec1bea7431d66c37953c3a848c1df37a1237f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.29, 0.33, 0.38], "teacher_probs": [0.29, 0.33, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.02734375, 1.248046875, 0.595703125], "student_probs": [0.16247114539146423, 0.5507073402404785, 0.2868214249610901], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.29, 0.33, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "818d017ed559d2e155f0798aba0f655ad78ec3c6fb0303c8624486dee8dc5982:action", "state_id": "9576f459758637ad76496d2e371c93f7114433a48a53c7a120226a4089a0efdd", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.3, 0.7], "teacher_probs": [0.3, 0.7], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.12890625, 1.126953125], "student_probs": [0.2693255841732025, 0.7306743860244751], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.3, 0.7], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5887d40e1f59c96bb43d8ee801fb93b697254989ae59546e95f4c0ffa8c2be99:action", "state_id": "a95e0bde4b7106808fbdc67202f1ffb063d0efd8d7e8767dcda8ea7cf3d580eb", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.5599999999999999, 0.14, 0.3], "teacher_probs": [0.5599999999999999, 0.14, 0.3], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.2265625, -0.03125, 0.6328125], "student_probs": [0.3054444193840027, 0.23602916300296783, 0.4585263729095459], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.5599999999999999, 0.14, 0.3], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "69c5952c048c3309dcbb6a3dd51fbbbb318c6d46ee2eb20a10649fe31c190f06:action", "state_id": "02db99841c7199b368b8508af6069474b8f72ac52c2c6dbfe64ebe78f87ee0c3", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.24, 0.12, 0.64], "teacher_probs": [0.24, 0.12, 0.64], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.63671875, -0.8271484375, 0.376953125], "student_probs": [0.21823076903820038, 0.18039041757583618, 0.6013787984848022], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.24, 0.12, 0.64], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ba6083c51cdbc3346b675072e778360bccd656a153d0fa5380e2b493ef891ed0:action", "state_id": "c92c06ce5650b93673a1cbf211340b1e4024a81728565b72e05d0457811b504f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.69, 0.31], "teacher_probs": [0.69, 0.31], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.7724609375, -0.650390625], "student_probs": [0.46952030062675476, 0.5304797887802124], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.69, 0.31], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a3d33f861580e145e6dab6f043befa713e28cb5a88140d3d98d75622bf44128f:action", "state_id": "23a6758af3dead2894d784aa2932aeb019edf0176aa76422f76aa07ddbbe5310", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.15, 0.14, 0.71], "teacher_probs": [0.15, 0.14, 0.71], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8759765625, -0.6273193359375, 0.23046875], "student_probs": [0.18846964836120605, 0.24167509377002716, 0.5698552131652832], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.15, 0.14, 0.71], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "629e0c2527a6b30a6c3a13b3ce21fa8b986ccb4511bc41aa1b0f48fc8accefe1:action", "state_id": "9480a56a8a379281d40fc6e60ad824161877dab73ff58320328e0886d3dc339c", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.12, 0.4, 0.48], "teacher_probs": [0.12, 0.4, 0.48], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.513671875, 0.24609375, 0.75], "student_probs": [0.1497865915298462, 0.3202100396156311, 0.5300033688545227], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.4, 0.48], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4ed621564a37af2d5d6fdb4b2e67f4ec8e1eed7ba9df7480631412ad328b3020:action", "state_id": "7e684a8d1dc5269463ebb0fff97ba2a7699bdd97788fa1fe62032e4ab8c3bbbc", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.12, 0.27, 0.61], "teacher_probs": [0.12, 0.27, 0.61], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.328125, 0.6640625, 0.763671875], "student_probs": [0.14977344870567322, 0.40395814180374146, 0.4462684392929077], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.27, 0.61], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "468d9d8d5f9fe798bbc043d6ddf741306c9edaab8a6358ff3a93bbae8c89d1e0:action", "state_id": "3a65119487295ca4fb847c7150c968563564c2c54c443a1e374ccf3e9c1a3e53", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.18, 0.82], "teacher_probs": [0.18, 0.82], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.27734375, 0.701171875], "student_probs": [0.27318641543388367, 0.7268136143684387], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.18, 0.82], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6360576da46e8b438ebf17addd9a20055f0ed295e2ff976f1e2ad810ab9b659a:action", "state_id": "dd2d6e2663140195b5c5cade92e1c95c57a069befed681335972057f0d262588", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.1, 0.84], "teacher_probs": [0.06, 0.1, 0.84], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.171875, 0.42578125, 1.208984375], "student_probs": [0.14714165031909943, 0.2674819231033325, 0.5853764414787292], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.1, 0.84], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b839f192398975b17915a1bdd86c5c86e7b534f1c49686982a2c8628520d3d31:action", "state_id": "6d5162d98e128c0abed277d9676c94957194dbf494ab77607a3c440d74f27d65", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.35, 0.65], "teacher_probs": [0.35, 0.65], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.12109375, 0.45703125], "student_probs": [0.3593641519546509, 0.6406359076499939], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.35, 0.65], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f381a7f8af8bcc05b4001466a172e912302ace3b69849e607b12d441e220a2d2:action", "state_id": "4b96f013ba01ed5379b0c9d501905f6091fbcf9b7db650a72c432aec3672cdb1", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.03, 0.93], "teacher_probs": [0.04, 0.03, 0.93], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.5550537109375, -1.2421875, 1.53125], "student_probs": [0.10462328791618347, 0.052627161145210266, 0.8427495956420898], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.03, 0.93], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0a70c890b9bfe2c711d224eabeea15a13b2478769f48b0903aac825079f8804e:action", "state_id": "b5e4fd318da8f3c69553ec8f84fdc9f73d7c41173f4c47e6223a13df5a7f7d5d", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.01, 0.94], "teacher_probs": [0.05, 0.01, 0.94], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.69921875, -1.1796875, 2.9296875], "student_probs": [0.09562986344099045, 0.014608140103518963, 0.8897619247436523], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.01, 0.94], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fd2e4b60dd57c4b9c4324bbea46e5b087f84f01826c614afb085e76ac20e8902:action", "state_id": "fbdd720ab445dcea5b1cb84ea3c724690429919f256a33fde27705029429563c", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.22, 0.37, 0.2, 0.21], "teacher_probs": [0.22, 0.37, 0.2, 0.21], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.25, 1.029296875, 0.9296875, 0.71875], "student_probs": [0.1481219232082367, 0.3228967487812042, 0.29228320717811584, 0.23669815063476562], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.22, 0.37, 0.2, 0.21], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "22646f3a75d89edd48da5bad8b7cbe5456b82ba5eadcda3c2e603805db77ba7c:action", "state_id": "59fa5f3a2f1e02feae03891f95f0a6aa9d358e62e620868ef46fe1511e1e529e", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.29, 0.24, 0.47], "teacher_probs": [0.29, 0.24, 0.47], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.25, 0.744140625, 0.546875], "student_probs": [0.2509576380252838, 0.41134193539619446, 0.3377004563808441], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.29, 0.24, 0.47], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fdd8fa923792555afc6fea2489171c0adcb3350154e81b21f5d41d8c52c082f7:action", "state_id": "f5618dadf3446ead3f622ff2c51e55e965ff965569cf2d17aa2deb218130f4be", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.48, 0.22, 0.3], "teacher_probs": [0.48, 0.22, 0.3], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.2265625, 0.078125, 0.20703125], "student_probs": [0.35177671909332275, 0.3032504916191101, 0.34497272968292236], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.48, 0.22, 0.3], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "57805cf4e4285c65c8a237c61308dc1bfa1d0e34ac5ea848443aebf1b78ade4a:action", "state_id": "dab26c02c227735705b7833da7e6230e068fdb0e8adf45cee01ab01748ff1927", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.57, 0.43], "teacher_probs": [0.57, 0.43], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.2890625, 0.30859375], "student_probs": [0.49511730670928955, 0.5048826336860657], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.57, 0.43], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d70acc46055074c86aea8580225546ec01e8d87d591d92f0b614c987a9a38865:action", "state_id": "2b80c2484375b9aa47a67f3a0dc14f4976fcadda5479b807b9b197fe6c305368", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.7, 0.16, 0.14], "teacher_probs": [0.7, 0.16, 0.14], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.69140625, 0.19140625, -0.57080078125], "student_probs": [0.5292239189147949, 0.32099053263664246, 0.14978554844856262], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.7, 0.16, 0.14], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "02ae75a7a2f8e1cfaae36e9638173661ad1ff2ce69561be498ee91ab396f2dfe:action", "state_id": "61daa48926ebd077b40279b210b454c8e95977b668d2c9663d4b945bd8b9998b", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.41, 0.49, 0.1], "teacher_probs": [0.41, 0.49, 0.1], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.751953125, 0.04296875, -0.8056640625], "student_probs": [0.5872744917869568, 0.2890234887599945, 0.12370196729898453], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.49, 0.1], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "06b3d5d41ea481662017e32084251fea8b979009fbdba35f4814e0476f30e74c:action", "state_id": "471b34c2c8725567a517da1934e375b241658597b1f47d48274b897663ef5a58", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.88, 0.09, 0.03], "teacher_probs": [0.88, 0.09, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.73046875, -0.6865234375, -1.2578125], "student_probs": [0.7249672412872314, 0.1757626086473465, 0.09927017986774445], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.88, 0.09, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "12c1fc0dd5b4f15c291fb45fe8cb254386fe0e3e025152ac597883b2d0f0569b:action", "state_id": "1e1ffae4ad1b6681aa4817356255b918fa6fe1b4e3dbd35382779ee6109d6872", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.58, 0.1, 0.32], "teacher_probs": [0.58, 0.1, 0.32], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.40234375, -0.54443359375, 0.658203125], "student_probs": [0.3731955289840698, 0.14479589462280273, 0.48200854659080505], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.58, 0.1, 0.32], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4299fec751dd65cb6af3b5f8e857239b86ee9ad93c20c562574c45868da5ae28:action", "state_id": "d761fcd5516b8e883514a7491d49f377fadba50f353573bc07d1859c3d952973", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.57, 0.1, 0.33], "teacher_probs": [0.57, 0.1, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.3515625, -0.4033203125, 0.771484375], "student_probs": [0.3342348635196686, 0.15711234509944916, 0.5086528658866882], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.57, 0.1, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "83b20d54c2b86c50df24a1bebaefd247cdf4e46f90351715e2b60873aabfdc4e:action", "state_id": "6334f40fc170ff96aa414adf5189bbd5d6daa89dbb8920ee5cce39d1375b45c1", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.62, 0.38], "teacher_probs": [0.62, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.6962890625, -0.716552734375], "student_probs": [0.5050657391548157, 0.49493423104286194], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.62, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "dc84d038e56941743ebba7ce87ada40e9ab048184d1ce1991b67413916f0253b:action", "state_id": "cf46ae7e393b46c1f897a9af84f00c5a32c23e12cd5e03da663682bbe112ebc5", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.15, 0.21, 0.64], "teacher_probs": [0.15, 0.21, 0.64], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.265625, -0.7880859375, 0.119140625], "student_probs": [0.15137773752212524, 0.24403637647628784, 0.6045859456062317], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.15, 0.21, 0.64], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8b1b7e9abb351150d6254b90a6ce499b78e258572d6a4894d5bac589e4f86ffc:action", "state_id": "59c411281930f7c3a47fcb34e68e703085645d742c9713af3fa16e76cc0bfb39", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.11, 0.3, 0.59], "teacher_probs": [0.11, 0.3, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.876953125, 0.05078125, 0.791015625], "student_probs": [0.11324820667505264, 0.2863790690898895, 0.6003727316856384], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.11, 0.3, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e07e71edc006c611ae071e4cff1703b9202ec445e1383fd5a5c393eb4fa721a9:action", "state_id": "fe8df469f8f190c0bae58f6be22a30e56761738823f8873af71e250ef5fb7459", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.29, 0.71], "teacher_probs": [0.29, 0.71], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.83984375, -0.103515625], "student_probs": [0.32380762696266174, 0.6761924028396606], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.29, 0.71], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "edf761c95f4fff5641d0d74b31af27bfc66123a8cd6540902b263e2f4fe6442f:action", "state_id": "e1bf332b09dda782efae8e15962e2b41bd4844fab48532d17be2ec9a03e349cf", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.46, 0.48], "teacher_probs": [0.06, 0.46, 0.48], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.974609375, 0.24609375, 0.486328125], "student_probs": [0.11494822800159454, 0.38962510228157043, 0.4954266846179962], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.46, 0.48], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "763efd424db205efc25ad2b6f9e55e5e266afd45786c4cb90a6d8d3af5f355a8:action", "state_id": "05fe78c34ffacd97d8fff19db03605a3230f0c3eff5162f09809352e7bafc117", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.17, 0.83], "teacher_probs": [0.17, 0.83], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8681640625, 0.224609375], "student_probs": [0.2510963976383209, 0.7489036321640015], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.17, 0.83], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e6ae2f2a388a84ea6f16b7d053af7da05a5f347d679e756e698891ec722e7e1d:action", "state_id": "7dd88d54cbd1e61a77b515255955b5c923c066091c09ff1f0463552a926d535c", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.09, 0.84], "teacher_probs": [0.07, 0.09, 0.84], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.50390625, -0.6685791015625, 0.986328125], "student_probs": [0.15907590091228485, 0.13492359220981598, 0.7060004472732544], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.09, 0.84], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e88bdaa1e7f7470ebf8c600cd434a45350b154e7e1a89fb0584bc8d7a40d9c90:action", "state_id": "3af47ca7e376fe3a0bf4f311d4f6c4708e76aea6c6e4a6b716843b41311e1984", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.46, 0.54], "teacher_probs": [0.46, 0.54], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.56982421875, -0.714599609375], "student_probs": [0.5361307859420776, 0.46386927366256714], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.46, 0.54], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "682f952309a523a2a55bfa2a9ebfa67419d9f04a41ec44daf6ac0ca48fa835e5:action", "state_id": "95b9a0f2b50eea9fa44da95e33faebc0e855f3c5f430daf265464750e68cf905", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.05, 0.89], "teacher_probs": [0.06, 0.05, 0.89], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.083984375, -1.4375, 1.08203125], "student_probs": [0.09591706842184067, 0.06735441088676453, 0.8367284536361694], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.05, 0.89], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d1789c82bd8102584b97ad466b793b4a7b7a5ff933a05e23af7e20a5eb0ac94f:action", "state_id": "f0a4aeae6cc22e4fc27eb3b9f5f452fbfedfb880bf574abedeb9ca433e3386af", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.52, 0.03, 0.45], "teacher_probs": [0.52, 0.03, 0.45], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.5546875, -1.4765625, 0.40625], "student_probs": [0.5016994476318359, 0.06580864638090134, 0.4324919581413269], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.52, 0.03, 0.45], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "428979a6bf8d945870785041317ddb79947d58e55e2e4ef64508d0fcfaeb7648:action", "state_id": "da9d609372e72ae9dc4766120b92bf15a36cd65d45fb7f9ca9792e1d572ab36f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.18, 0.82], "teacher_probs": [0.18, 0.82], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.44140625, 0.259765625], "student_probs": [0.15431226789951324, 0.8456876873970032], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.18, 0.82], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c4f187b53995d7619092bff91a29a8e8d3d7aefdb85db9d3770f6bffd572d4bb:action", "state_id": "4691cef86fdb63aca6d26c4196f626882f39221557a369566e5ba789c834c764", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.4, 0.03, 0.57], "teacher_probs": [0.4, 0.03, 0.57], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.6484375, -1.296875, 0.748046875], "student_probs": [0.44490283727645874, 0.06359554827213287, 0.49150165915489197], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.4, 0.03, 0.57], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4856d809da9c10a6ab30f665638c01d529769201a2cab7f2b9883130b0ec3bd2:action", "state_id": "6b73b50116bec2a42f1894a69840161e676436b6a11efd1a12fc5e07e60fa32a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.0, 0.95], "teacher_probs": [0.05, 0.0, 0.95], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.416015625, -1.29296875, 2.9921875], "student_probs": [0.06979455798864365, 0.012636275961995125, 0.9175691604614258], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.0, 0.95], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e233c68a5a9ef6fd7f2bd7ebe49384cdb50948ec893eb934343ea98b9d0dc8f3:action", "state_id": "a1d06a571c89fef259821442134813fe6061e7409b68a70eda8c3b528ba07d9f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.36, 0.03, 0.57], "teacher_probs": [0.04, 0.36, 0.03, 0.57], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.115234375, 1.064453125, -0.818359375, 1.181640625], "student_probs": [0.1189592257142067, 0.3870168924331665, 0.05888908728957176, 0.43513476848602295], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.36, 0.03, 0.57], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4ce483378c5751431f342d4d9c6fc0b8b74b0fb6e2a584a941959dc74dc35a13:action", "state_id": "9ac22992e16b634733a960b2677ce01642e719363ab676e5b712bcb392c8c71d", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.63, 0.1, 0.27], "teacher_probs": [0.63, 0.1, 0.27], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.513671875, -0.69775390625, 0.3125], "student_probs": [0.4726915955543518, 0.14075452089309692, 0.38655388355255127], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.63, 0.1, 0.27], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "158e682f5252346c86a79772aecbe37490daa1b5e45e237ffad015648406180f:action", "state_id": "b59b9808f1f1088c38ddddf7b530316cb150a10b4a73289518154ed6038031ca", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.49, 0.51], "teacher_probs": [0.49, 0.51], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.7998046875, 0.078125], "student_probs": [0.2936069965362549, 0.7063930630683899], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.51], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cb025a0558dcef132429f386455d4bf9a47995f30489385d9e397978fcfab1c9:action", "state_id": "6cab4a26d0f03a2dde223e617c3e3dd45b4f3148c194dfb17c6477f9b3c436ac", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.41, 0.2, 0.39], "teacher_probs": [0.41, 0.2, 0.39], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.46484375, -0.74560546875, 0.03125], "student_probs": [0.5138115286827087, 0.15314839780330658, 0.3330400586128235], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.2, 0.39], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "47084532a1da077ae39a575792ca741b811f9ba7f18e4adc556eda6d6c932e1b:action", "state_id": "5cd5e2825d686bc459d6e18de830a684dad4c9efc2f5b9cc44a9519841fd0e11", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.24, 0.52, 0.24], "teacher_probs": [0.24, 0.52, 0.24], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.751953125, 0.41796875, 0.28125], "student_probs": [0.42723578214645386, 0.3059285581111908, 0.26683565974235535], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.24, 0.52, 0.24], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6e3e188471d2546d5529ffa03f77a8196d33a5a7d43146798ac0d00bf868ce3c:action", "state_id": "54324d7d2a791d84397380fb456395fb1c125ab6025ff9101d67541389a84cc8", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.15, 0.59, 0.26], "teacher_probs": [0.15, 0.59, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.6015625, 0.734375, 0.26171875], "student_probs": [0.3503955900669098, 0.40016433596611023, 0.24944016337394714], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.15, 0.59, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f1eb63dc0c58ee03c23cc34e1a5204147285d761d4ae3004df4758f870e36248:action", "state_id": "783b82fd1c1cfa806241157192e43f9cc025495b04e58410730f0f0b3f509fad", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "west"], "gold_index": null, "teacher_raw_probs": [0.34, 0.66], "teacher_probs": [0.34, 0.66], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.62109375, 0.32421875], "student_probs": [0.5736784338951111, 0.4263215959072113], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.34, 0.66], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8fd814c480e9da4dc8c005bd74dac0d52db2fa04a1296757ff74571595cd90db:action", "state_id": "74c231da4d5aaf6b87f59bb08203cbf1241694cd7a2d383df173a46d593221f1", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.5, 0.28, 0.22], "teacher_probs": [0.5, 0.28, 0.22], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.27734375, 0.37109375, -0.265625], "student_probs": [0.37323102355003357, 0.4099140763282776, 0.2168549746274948], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.5, 0.28, 0.22], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "31600796839cdbd5a22721cfb888492dfd9aa13fb53e9cb82ac78bc1d4100486:action", "state_id": "86b02bbaad340b426b7bce60097212c4f8d9b49b58cdb6bfdfdf05b0ed7d172a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.67, 0.33], "teacher_probs": [0.67, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.27734375, -0.357421875], "student_probs": [0.6535692811012268, 0.3464307487010956], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.67, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7aaa4bf2f3f3718c35f6a5875abf8ea90d78bef204f23f681b108b23de5cbfa7:action", "state_id": "0acafc11547ae0c0d867b7f0f782c8ecebc1c7dbcab27f38fb6ccdb8f35c5c21", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.43, 0.4, 0.17], "teacher_probs": [0.43, 0.4, 0.17], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.1328125, -0.119140625, -0.8447265625], "student_probs": [0.399286150932312, 0.40478262305259705, 0.1959313303232193], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.43, 0.4, 0.17], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b76237e1f1f776b312f694c94c75b6b71ab7f1b18d3ab2f1283724d68a6b24e9:action", "state_id": "6582d48822477f1684724974ee627865422cfa526ef9e86939652dd58470e24c", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.57, 0.31, 0.12], "teacher_probs": [0.57, 0.31, 0.12], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.5703125, -0.2578125, -1.017578125], "student_probs": [0.6093013882637024, 0.2661840617656708, 0.12451452761888504], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.57, 0.31, 0.12], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b71517da9f5b060d1aaf776043a8bc6ef1509eab8d9e030d41939d5e808db0ff:action", "state_id": "f168f8aa7e70b42da23733186bf9132d5f4c1e4f31269045c0bb6c0fb51cd865", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.65, 0.35], "teacher_probs": [0.65, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.427734375, -1.046875], "student_probs": [0.6500231027603149, 0.34997692704200745], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.65, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "faec8887cb1544872b85807e417efb54a96f889e70610224c24dbab99f814c94:action", "state_id": "753998b54934713d9e5d7809f1ca8f4a80ddd3ad982ec63a1bd65a9749a86f50", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.65, 0.18, 0.17], "teacher_probs": [0.65, 0.18, 0.17], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.361328125, -0.818359375, -1.5234375], "student_probs": [0.6852885484695435, 0.21064041554927826, 0.10407110303640366], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.65, 0.18, 0.17], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b77b61e9bf6766b8a237045611bd1c6e12fdbbbebd316c0e0a65959b9b45f907:action", "state_id": "9e3e72c255d04cf3dacec0909fa0852617aa0155a75c3a609264d8517e80d50c", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.55, 0.1, 0.35], "teacher_probs": [0.55, 0.1, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.97265625, -0.72412109375, 0.53125], "student_probs": [0.5475237965583801, 0.10034643858671188, 0.352129727602005], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.55, 0.1, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "70b1551f65ad0ef73d4bcd11f5e50637e70e5be8646a9e31ccdaeccf96ab1de8:action", "state_id": "4dd5459943a450e035bec42eec21fb598fbcee53e280b305079fe9fc0d567416", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.38, 0.17, 0.45], "teacher_probs": [0.38, 0.17, 0.45], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.271484375, -0.302734375, 0.94921875], "student_probs": [0.5176854133605957, 0.10724854469299316, 0.37506604194641113], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.38, 0.17, 0.45], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5394d548791667ead1a272492a4e2e176f0c922e5056fd72fb4795a110c99826:action", "state_id": "2a9c75d3b1ba1f3081fbc10071a1981880cc95447d8930bcb9918c11a009f3e2", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.75, 0.25], "teacher_probs": [0.75, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.150390625, -0.283203125], "student_probs": [0.8074606657028198, 0.19253936409950256], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.75, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "274b4933658eea5bb96bc122301d2e91472a58dbb2337cb2aba355e58da05b90:action", "state_id": "e25cf931ff28291be3f5502affbfe13933450211061b70595350ec65cff66504", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.33, 0.08, 0.59], "teacher_probs": [0.33, 0.08, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.16015625, -0.494140625, 0.794921875], "student_probs": [0.29356613755226135, 0.15259785950183868, 0.5538359880447388], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.33, 0.08, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "98574c40866c3fd3ab3da9d12083471167af17508ec4aa7cd6cc2af04aff047a:action", "state_id": "95ef2a6876a03203c39aedee234662e7e17ad0904445124e9a35a920b923c47f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.78, 0.22], "teacher_probs": [0.78, 0.22], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.140625, -0.4658203125], "student_probs": [0.6471295356750488, 0.35287052392959595], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.78, 0.22], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4851345aa4cb60d3a52c411e27e5b5b6d3bf7dcb554dfd1c955e81f53a72f56c:action", "state_id": "9e3138afb7d9cedec3e83a0f632fa7ee10e76bfa525b44356cece5453c4f97d7", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.09, 0.09, 0.82], "teacher_probs": [0.09, 0.09, 0.82], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.66302490234375, -0.599853515625, 0.818359375], "student_probs": [0.1546972095966339, 0.16478492319583893, 0.680517852306366], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.09, 0.09, 0.82], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fc60a8958c5c197e09836e9cb0b13d093271bb16703409d5f692de6bbbc53175:action", "state_id": "5ba7d6af5114af084dc0dfe2a7adeef12a04411f2bd710d75489e436328133a9", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.95, 0.02], "teacher_probs": [0.03, 0.95, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.314453125, 2.9609375, 0.587890625], "student_probs": [0.03342365473508835, 0.8841745853424072, 0.0824018269777298], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.03, 0.95, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bc9d0752a117c3e61ba7618b3e5f27ff23c8844a43b30bcf75efa592a6aebd84:action", "state_id": "4669b33fcc312767b786414f1fe5b7977bc036d494bb6c421cb7324eca2c9201", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.23, 0.06, 0.65], "teacher_probs": [0.06, 0.23, 0.06, 0.65], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.134765625, 0.99609375, -0.8740234375, 1.12890625], "student_probs": [0.12324110418558121, 0.3818401098251343, 0.0588436983525753, 0.4360750913619995], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.23, 0.06, 0.65], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4a1ee033e85c6c6c826d8ae5e76bd03c5a70dc1c25367ef72eb47a4e5265c1b6:action", "state_id": "daa92c4ef96b55c96f215827917adacb68ed0d6a6cb41c26c1599466dfca2b70", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.42, 0.19, 0.39], "teacher_probs": [0.42, 0.19, 0.39], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.6484375, -0.53271484375, 0.46484375], "student_probs": [0.46746474504470825, 0.14347654581069946, 0.3890586793422699], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.42, 0.19, 0.39], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "01a2d98ac4c874548a0764dc436719493e3d0f885900352f37645b546f01c33b:action", "state_id": "c2232e71ad7b7fb0ec1e839a5f6a12b7d6c97087d791642200eadff2c08c819f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.44, 0.56], "teacher_probs": [0.44, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.67724609375, 0.2265625], "student_probs": [0.28826844692230225, 0.7117315530776978], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.44, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a4d6cad45e2938cad46af059502edc54b11cc1edec2a245bbe9fec02ab8c25c7:action", "state_id": "8415d11f0132fb100683e88e1f64f8e23bdb5097cba9a64085fd69dc0cc9082a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.34, 0.18, 0.48], "teacher_probs": [0.34, 0.18, 0.48], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.751953125, -0.5654296875, 0.341796875], "student_probs": [0.5177639722824097, 0.1386754959821701, 0.34356051683425903], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.34, 0.18, 0.48], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "799eab8545fbe80d68b74b43c9de8b1980ced9f796a09e99c61d1dab6c4d340a:action", "state_id": "5e29fc70a93e90b10aba53a7f07a3de662f501a588f16b69fc1cf2e8067d8fd5", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.55, 0.11, 0.33], "teacher_probs": null, "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": null, "student_logits": [-0.216796875, -1.001953125, -0.7333984375], "student_probs": [0.4871886372566223, 0.22218161821365356, 0.29062968492507935], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": "Rounded target has no identity probability-simplex representative", "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": null, "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "29fdcb6c85ca527ff2b7ba1bffb414ea183ae908c143a5cb47a21c90ae86fe24:action", "state_id": "d964d51bb17880ec0efbf35127585d2a5279e59ab7c4723fba4c6fbfcab59ad6", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.49, 0.51], "teacher_probs": [0.49, 0.51], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.78271484375, -0.683837890625], "student_probs": [0.4753008484840393, 0.5246990919113159], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.51], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f6cfd06c97570db0d84f2ec7fea1a8370a2fac21ddc8b507f410e6d32c4e766a:action", "state_id": "fa7b2b59a7bc4087843650a82930f1ae3c6f76dd30580b734f62e1e62b0df59d", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.5, 0.21, 0.29], "teacher_probs": [0.5, 0.21, 0.29], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.00390625, -1.03125, -0.55712890625], "student_probs": [0.5173171758651733, 0.1851770579814911, 0.2975057363510132], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.5, 0.21, 0.29], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "952ba1731b24e31552faf4933964663d82143f25dc3fb0b415c08f314b497339:action", "state_id": "631c538a0d91ca0c16bd231b6c395458b0fd369c9a2fb756fd5c77d429a9f500", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.43, 0.51, 0.06], "teacher_probs": [0.43, 0.51, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.4609375, 0.3125, -0.5703125], "student_probs": [0.45073169469833374, 0.38855499029159546, 0.160713329911232], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.43, 0.51, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "46422a4bdd450aa230d3ab9b8874625442238ac8579847bb894e0db62ce659cf:action", "state_id": "c1add2e8d2b228b29fe281890a1ace14123fd4a43848f9e0b562017f353fc204", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.58, 0.39, 0.03], "teacher_probs": [0.58, 0.39, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.326171875, 0.1640625, -1.00390625], "student_probs": [0.47285687923431396, 0.40209299325942993, 0.1250501275062561], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.58, 0.39, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bd80cf4abc3641c58d6fcdcf7a4f17038ff298ef4c212234f7a3be2c30506cc0:action", "state_id": "60fd37594cedcf7fdf0949875a09379f2e3974d8c0d4f63a7fad08a1e3c12e84", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.33, 0.44, 0.23], "teacher_probs": [0.33, 0.44, 0.23], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.34375, 0.1796875, 0.25], "student_probs": [0.36242398619651794, 0.3075852394104004, 0.32999080419540405], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.33, 0.44, 0.23], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "74b98d67f6f0f21034909e8484c75092d74abd9ff788ebe965ec3a5612187ffc:action", "state_id": "762bd6439bd585fb084db0020461293309b2cbd78f066e86aa6a7698c7a0f8d5", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.47, 0.53], "teacher_probs": [0.47, 0.53], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.44921875, 0.19140625], "student_probs": [0.5640984773635864, 0.4359015226364136], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.47, 0.53], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4d1afe7610d87a194042624be1fa343016b2fb41943d518066bc16696a9c17c3:action", "state_id": "549f1b51ebee40a823799662e5da9cbb7fd413ec15d347c5e57c461fd4641990", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.36, 0.39, 0.25], "teacher_probs": [0.36, 0.39, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.5234375, 0.6015625, 0.736328125], "student_probs": [0.30134034156799316, 0.32582658529281616, 0.37283313274383545], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.36, 0.39, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "490c72b8ce2f735fdeae2fa727b5fd5a6ac20bbfd99d76ec26d7ce14635799fb:action", "state_id": "5870856638d048419d67c94ebe501fc880274507e64d77e17b57caa73222fe4e", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.41, 0.59], "teacher_probs": [0.41, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.51953125, 0.64453125], "student_probs": [0.4687906503677368, 0.531209409236908], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b2dca1580573c0eb95d1b829db007636c77725ba5faee7237758837b3ff899d0:action", "state_id": "4ed6a9f0291dfc3d2ba2e2f5dc68f438b4adcb30f627edd0e34df78b0a66cba8", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.24, 0.35, 0.41], "teacher_probs": [0.24, 0.35, 0.41], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.490234375, 0.59375, 0.140625], "student_probs": [0.3553626835346222, 0.39411965012550354, 0.25051769614219666], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.24, 0.35, 0.41], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b47592a5acdb739dd4f083ebe17b21fbc581cb814d1c78b045258b85bff07df6:action", "state_id": "1295fb504ab624fb29aab2fa204cc2a85d8a8a1d00e2ffbbeb520de3f91a008a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.44, 0.56], "teacher_probs": [0.44, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.48828125, 0.46875], "student_probs": [0.5048826336860657, 0.49511730670928955], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.44, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e38c85cb9e7b5d8dada6f81291ce5ba3b4ca672b3d740f999eaf61c524afad49:action", "state_id": "3d47a8e34377c2551708d3d1fa748335245c2d28aef56da48073b0e8e58d8c94", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.68, 0.13, 0.19], "teacher_probs": [0.68, 0.13, 0.19], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.7734375, -0.59619140625, -0.40625], "student_probs": [0.6403786540031433, 0.16278506815433502, 0.19683624804019928], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.68, 0.13, 0.19], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fb64bec1e1e559e5a3fe531ba38b30dff4b5de311415a993df15d9603581637b:action", "state_id": "87d1e78367624ab6b40e8f0697f51066a3724f2ada348f188b1c60545203cc34", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.35, 0.13, 0.52], "teacher_probs": [0.35, 0.13, 0.52], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.0078125, -0.4638671875, 0.69921875], "student_probs": [0.276206910610199, 0.17234021425247192, 0.5514529347419739], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.35, 0.13, 0.52], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6a7bccd12be1cfdd1f47f17a3309ba032a42fba0e6489bb7d94bf2829b7519a0:action", "state_id": "17417e8b35043046b3c2ab20a2f6b162e746cceca350472c110518c469ddcb1f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.67, 0.33], "teacher_probs": [0.67, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.015625, -0.39453125], "student_probs": [0.5936092734336853, 0.4063906967639923], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.67, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ea4f1abf0820eb833a87063ae6391244db8182754ecff12aa8b6c829006e3a57:action", "state_id": "7fb912c6a3d68e374f922a3510659d18543a5a347e94a0fabc5432c85b376b55", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.12, 0.14, 0.74], "teacher_probs": [0.12, 0.14, 0.74], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.775390625, -0.5595703125, 0.525390625], "student_probs": [0.16911762952804565, 0.2098546028137207, 0.6210277676582336], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.14, 0.74], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a63fdfbeddc045826558a45710a8b72642a10a49f322109cb279bba8edd179c4:action", "state_id": "17865b0b62b1420304c8e30ebd260e1f1b0fc74b5e931bc183c207eb70ed6828", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.94, 0.03], "teacher_probs": [0.03, 0.94, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.5263671875, 2.673828125, 0.41796875], "student_probs": [0.03557652235031128, 0.8729525804519653, 0.09147099405527115], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.03, 0.94, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3e591aa7d166c2057e18cfd726a1a35e2e7ad24a78d5bf313a10c5b949d399ad:action", "state_id": "9b606726eee2f5878cca9fe0032daac1cca43cf9ca7ce6f6a4e99c3044dec8ca", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.6, 0.38, 0.02], "teacher_probs": [0.6, 0.38, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.154296875, 1.26171875, -1.61328125], "student_probs": [0.459512859582901, 0.5116233825683594, 0.02886381559073925], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.6, 0.38, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5923e2168ed6ac17296580f8ceb8a9b87c1c419ba1fd1f8acbe668be64dcf998:action", "state_id": "3f3decc8d6401dd4d35d0687bcd09b88ff4a9e8463b2f0b60e6589b83df69fa0", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.45, 0.51, 0.04], "teacher_probs": [0.45, 0.51, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.427734375, 1.091796875, -1.31640625], "student_probs": [0.3207736015319824, 0.623156726360321, 0.056069664657115936], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.45, 0.51, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c482da6954de58c4fd85ec10fe457e0bfe830b7ccb97dad672baead37a388e2f:action", "state_id": "4e2a816b3cfd1f2e0e9b2f3a82d6c2eff2fedc00a52d7ba967bbb3c9026f45be", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.75, 0.2], "teacher_probs": [0.05, 0.75, 0.2], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.625, 1.251953125, 0.162109375], "student_probs": [0.04043305292725563, 0.7180941700935364, 0.24147282540798187], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.75, 0.2], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c3ece9b09212769299140b6531c6f01ac25ba20a1246b30abaa852a8563937fb:action", "state_id": "b694659be255639abc25e2bd3b80e8dc9fb5c08bd4199eccdba0fa24034f562d", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.64, 0.31], "teacher_probs": [0.05, 0.64, 0.31], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.25390625, 1.337890625, 0.80859375], "student_probs": [0.045005809515714645, 0.6009960770606995, 0.3539980947971344], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.64, 0.31], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "47f2e4f5e3f0d99784e04ec69488abf84b12fcc8c8fe66b4bfc8d5df078d9692:action", "state_id": "8dddd163b19bfb30aca9b4af3db505d9fb0b2b532ca2bf7a8e22b4ad223446c9", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.46, 0.51], "teacher_probs": [0.03, 0.46, 0.51], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.1171875, 1.08203125, 1.376953125], "student_probs": [0.045188985764980316, 0.40751272439956665, 0.5472983121871948], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.03, 0.46, 0.51], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "016977844c7d2cdd7eb516588ecc6ed737b37308c95d9b898c62fa03945e0e1e:action", "state_id": "cd333be2ce5cf324e407d3a7ae6d84ece7bb217027a9c63d618cbcfea4451de0", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.27, 0.02, 0.71], "teacher_probs": [0.27, 0.02, 0.71], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.453125, 0.24609375, 2.4716796875], "student_probs": [0.46975818276405334, 0.05168599262833595, 0.4785557687282562], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.27, 0.02, 0.71], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3c2fd5f89d9cbfc801061e4c3eb6b4e085e79befeec2a03f4bf3a80d4fdab9ce:action", "state_id": "bb9ab269b46c50a179e156db5eff4e4cae30a2dda662cd3d1c0ca7fa8f02c036", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.01, 0.98], "teacher_probs": [0.01, 0.01, 0.98], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.134765625, -1.6875, 3.14453125], "student_probs": [0.03601168841123581, 0.007622536271810532, 0.956365704536438], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.01, 0.98], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9090f43e0e78f8ded490b3715079403aa5971dde9712559b048ff3fb5b7e3dcd:action", "state_id": "997e7a5c10b1e4bb5cc8e04a218f77c6c8771a3db13d496e1679f07be28c2a8d", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.6900000000000001, 0.29], "teacher_probs": [0.02, 0.6900000000000001, 0.29], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.6875, -0.05078125, -0.755859375], "student_probs": [0.04572867229580879, 0.6387059688568115, 0.31556543707847595], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.6900000000000001, 0.29], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "63e84db2c32e829707db4feab3eede0460b8c7ecdad688451be3fca37e1528ea:action", "state_id": "4af36e9d2f7ab2b6dad8524be2bc2fa8cd58ea5bb0d746d9405d7443944e5722", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.15, 0.73, 0.12], "teacher_probs": [0.15, 0.73, 0.12], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.2890625, 1.064453125, -1.41015625], "student_probs": [0.19242115318775177, 0.7448643445968628, 0.06271450221538544], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.15, 0.73, 0.12], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4eb27e2de56ddf53ec6c1f1f620436eaf69ff9a859132161135670080a94d3bd:action", "state_id": "5f472f55f22d208bb04577836e75631328e16bf6df574429a2944396683b57ce", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.92, 0.03], "teacher_probs": [0.05, 0.92, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.04296875, 1.447265625, -1.29296875], "student_probs": [0.17468346655368805, 0.7752689123153687, 0.050047650933265686], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.92, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "808793cc2fd34155818eb91ab2c9854127f301dd116a81c0f09f930712be7e5f:action", "state_id": "a8382528d11816058f23ab4090c69fbd6ada689f40804f76b722000fb3bf5d7d", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.99, 0.0], "teacher_probs": [0.01, 0.99, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.548828125, 3.8203125, -0.7548828125], "student_probs": [0.03620309755206108, 0.9539669156074524, 0.009829948656260967], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.99, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "055ec13a8740aa40eb4d5d6a14958035ebb320a80de6dc19763036c093f2acb9:action", "state_id": "37063b684f3ffddf781023f996683efc8d644128f7e509e660bfcef7e956c64d", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.5599999999999999, 0.41, 0.03], "teacher_probs": [0.5599999999999999, 0.41, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.3359375, 1.044921875, -1.40625], "student_probs": [0.5518966913223267, 0.41254499554634094, 0.03555829077959061], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.5599999999999999, 0.41, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9b9f1d6c34b7eb1613b09f979dbbeee8d4a980d0472c7d2f94ecbdd0a7c94857:action", "state_id": "68fb3145ef4b36c2ac3f2b8e5976d9915ae4ca75bde6c21d92579b601ca2d981", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.45, 0.49, 0.06], "teacher_probs": [0.45, 0.49, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.462890625, 0.935546875, -1.30078125], "student_probs": [0.36027413606643677, 0.5779697299003601, 0.06175613775849342], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.45, 0.49, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "92490fd44cc48379a09a101bbcd7f6e1c454594ec7c72a0c2c8401d59106cbc4:action", "state_id": "4b86113a2bf9841714cb04765f8bbe46df5a70b2bc05382a8d573c34aa481a61", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.61, 0.33], "teacher_probs": [0.06, 0.61, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.51171875, 1.05078125, 0.47265625], "student_probs": [0.04707500338554382, 0.6104779243469238, 0.34244707226753235], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.61, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5d4e539968d10edd6c5f4df3d5617f64ccd821b6157f098c8d7df9eac99c444e:action", "state_id": "75dfd01d4086119ba3c58ffafb5a815ee94b94d2482f6a7c268034dee91b26d9", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.43, 0.45, 0.12], "teacher_probs": [0.43, 0.45, 0.12], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.27734375, 0.6796875, -0.74267578125], "student_probs": [0.3501507043838501, 0.5235891938209534, 0.12626010179519653], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.43, 0.45, 0.12], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ca3886cdc16cd2779b99e4c36878c310066a152371bc70a5ba74c5e4f5cfcdf7:action", "state_id": "c3ecc5c225c9267e587e90bc1f9614b8609e902a804f003d49c945b25c4c36de", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.46, 0.49], "teacher_probs": [0.05, 0.46, 0.49], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.099609375, 0.357421875, 0.5625], "student_probs": [0.09466459602117538, 0.40641355514526367, 0.49892181158065796], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.46, 0.49], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e23daa9a4091974462817e21a74c246456ed071f5b5f52a93330cd0e0181e391:action", "state_id": "597efb5335d6ce2d31ccf151d6b73805e9ec7bb8494ef8eddd7de767c7c0ae87", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.22, 0.05, 0.73], "teacher_probs": [0.22, 0.05, 0.73], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.404296875, -0.3720703125, 2.208984375], "student_probs": [0.29366424679756165, 0.04970322921872139, 0.6566325426101685], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.22, 0.05, 0.73], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d866c66eafc2177d1a2fa543bd0322846679083a47a6069303a1dd1679f162bf:action", "state_id": "8cb3004d034d50e50289e0d9ce502b544b6becade68bc3f657f400e16b9dfd33", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.37, 0.04, 0.59], "teacher_probs": [0.37, 0.04, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.2041015625, 0.18359375, 2.396484375], "student_probs": [0.4264896810054779, 0.056547462940216064, 0.5169628262519836], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.37, 0.04, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "604841425991d2880b04d98d215dc4fd5f4134da85d8af2e23c27d1d857a2cf6:action", "state_id": "7b667d8311be3021d2931d393e48bd4c72a2be374e6c781db959db11f68f0548", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.89, 0.02, 0.09], "teacher_probs": [0.89, 0.02, 0.09], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [4.044921875, 0.0078125, 1.40234375], "student_probs": [0.9184204339981079, 0.01620866358280182, 0.06537089496850967], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.89, 0.02, 0.09], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "51fa4acd1210e813f0fb791c556e6a5b71509393af3f36b84f9be98917212944:action", "state_id": "6a68b6d7e6e14c1bbc282122ce330848ebff19725fd111b9c34f157a94717eb6", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.21, 0.43, 0.36], "teacher_probs": [0.21, 0.43, 0.36], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.76904296875, 0.1640625, 0.12890625], "student_probs": [0.16675125062465668, 0.4239470958709717, 0.40930166840553284], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.21, 0.43, 0.36], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fcadf1bc84bf6a94d4dec1c64caa39d8afc2c6b6377e0a196969cce150d6fa18:action", "state_id": "4a4a576e6466768227e1b5ed0b73586e89ee1dc09ba115d50ff53e06ce9c83f3", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.51, 0.1, 0.39], "teacher_probs": [0.51, 0.1, 0.39], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.845703125, -1.46875, 0.01953125], "student_probs": [0.2556321620941162, 0.1370975375175476, 0.6072702407836914], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.51, 0.1, 0.39], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4f1b572c60d1eee195f9dfcdeee63bd2808f01cac64574461bbb3bc3751c68ac:action", "state_id": "cec63e6dc09790aa1563b27ecb352d68707de649f62a3454b7aafd78ff9aec54", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.48, 0.04, 0.48], "teacher_probs": [0.48, 0.04, 0.48], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.423828125, -2.40625, -0.5085067749023438], "student_probs": [0.48625293374061584, 0.06697417050600052, 0.44677284359931946], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.48, 0.04, 0.48], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4654681111e15fd89c226752635fc59d1dcda56572319d8a8fdba7b186c90081:action", "state_id": "1e3d6d41d409c78bea9abe069a20f8d2a9aa080777d2a18df960b0e23cf988d2", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.48, 0.04, 0.48], "teacher_probs": [0.48, 0.04, 0.48], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.58984375, -2.515625, -0.900390625], "student_probs": [0.5322524309158325, 0.07758209109306335, 0.3901654779911041], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.48, 0.04, 0.48], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "33c9854abb21e7cc5a6fc036df421f214a95bcb4a2b2af4a95d692ba3f0ba424:action", "state_id": "fafde744bb0b503820eaef4126aaf5197eb115de76ac6c5d78d511f8245d8076", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.3, 0.02, 0.68], "teacher_probs": [0.3, 0.02, 0.68], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.84765625, -2.5, -0.953125], "student_probs": [0.47812509536743164, 0.0916089192032814, 0.43026602268218994], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.3, 0.02, 0.68], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cd952f0ca3d60e70266965de8be84a2ed06172ab5b94ea518a3f2def2d2f2690:action", "state_id": "c6f1aa8e7f9a3ff6a3a6f04dae2ee76032898fecd640a5576ba90da006cf04e3", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.11, 0.48, 0.41], "teacher_probs": [0.11, 0.48, 0.41], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.6484375, 0.22265625, -0.3203125], "student_probs": [0.08873619884252548, 0.5763768553733826, 0.33488693833351135], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.11, 0.48, 0.41], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f09ba1f0641025e9170e42d3461938a32c3d24b522fa4f7f5d6dc50d53da5a70:action", "state_id": "a8a9b89b50b58a37527fe0702c5e48285f640989899de34c4bb00365d99eb9d8", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.51, 0.43], "teacher_probs": [0.06, 0.51, 0.43], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.36328125, 0.59375, -0.134765625], "student_probs": [0.08699861168861389, 0.615800678730011, 0.2972007095813751], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.51, 0.43], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "48d175ccd7d2c398b178995ecbd1d352ee321d7e8d18d96aedad8e4a5d8737ee:action", "state_id": "aa4ddc5cf6c5669eb8eabb2fe61ee4d122e3816c9140b082c06471c1df04cb49", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.56, 0.39], "teacher_probs": [0.05, 0.56, 0.39], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.080078125, 0.962890625, 0.4296875], "student_probs": [0.07553358376026154, 0.5826264023780823, 0.34184008836746216], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.56, 0.39], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "20a9d84d6154c55a2eab2c5bd7c35f07405bfb131903fd8d4c39f387cf4b4a02:action", "state_id": "03a35895c898fbe70dfcc55c0a4019e6bce65e6d59d02379740ac25c3f4492c7", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.08, 0.59, 0.33], "teacher_probs": [0.08, 0.59, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.04296875, 1.037109375, 0.53515625], "student_probs": [0.07219718396663666, 0.5779452919960022, 0.34985753893852234], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.08, 0.59, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2e7c7687c7efb40ae91b22a3716efd7914650343efbaafd4bc0032bc353a8faf:action", "state_id": "2e2d635b9e330813f14d4c7fceb0dde470475e9f34b3b986aba7e0aace678e1a", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.6, 0.33], "teacher_probs": [0.07, 0.6, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.1328125, 0.962890625, 0.6953125], "student_probs": [0.06513229757547379, 0.5296009182929993, 0.40526679158210754], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.6, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "880c46a181f83cf079066fab8c32352de44f5299d2fe1686c1eafb50b294cc94:action", "state_id": "7a83b918a2ecd71fef2ed3bd887ec06b923046b483a959826e8fa95678280eb9", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.44, 0.53], "teacher_probs": [0.03, 0.44, 0.53], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.40234375, 0.642578125, 0.984375], "student_probs": [0.051004018634557724, 0.3941873610019684, 0.5548086166381836], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.03, 0.44, 0.53], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5db251450def53e74ad38b67451edca7e5260272a33744fa1b219c2502feaf61:action", "state_id": "55d42c23e90010feac917b1bcc43fb0c6ac191aeb219ba3d8f71640e2f196d5e", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.07, 0.86], "teacher_probs": [0.07, 0.07, 0.86], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.05078125, -0.908203125, 2.416015625], "student_probs": [0.08312679827213287, 0.03186099976301193, 0.8850122690200806], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.07, 0.86], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c4b27e4f1412f65dad0724027c46f9eef96612dcb21c06cc3e59151c1f909a5d:action", "state_id": "f8934c230eba78c41a878f9b273fbb3dd74c0e409837e42082436ac511ef8162", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.05, 0.89], "teacher_probs": [0.06, 0.05, 0.89], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.3203125, -0.865234375, 2.5625], "student_probs": [0.09328809380531311, 0.02850688435137272, 0.8782050013542175], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.05, 0.89], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c1e4f8c1eea4cda70165d39d2985c03db09ac9751dc600727e42a6866cfd1b53:action", "state_id": "e55927dfcf702d29364209f18f9631eba5f12ac415910eb4bfe0bc9d0e827cd5", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.03, 0.91], "teacher_probs": [0.06, 0.03, 0.91], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.7109375, -0.51318359375, 2.53125], "student_probs": [0.13390818238258362, 0.039371147751808167, 0.826720654964447], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.03, 0.91], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "816c5eeb83c4aade75668cc5fdc9e4e9840356260c1425b99c152b7467d2c7f6:action", "state_id": "a00fe790f30c8ac233d1dac5b46846c370899a6b818cdebdaa1126db37e3aa7d", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.02, 0.89], "teacher_probs": [0.09, 0.02, 0.89], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.51953125, -0.681884765625, 2.7890625], "student_probs": [0.09111092239618301, 0.027403250336647034, 0.8814858198165894], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.09, 0.02, 0.89], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "45e3663678af2bb47af8e833ce844858796927be3dc85b53ace03aa5ac106264:action", "state_id": "d23a6af2a73f4a641b61615d3757d1010a64110bbab7cd3c18587251d908d290", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.01, 0.97], "teacher_probs": [0.02, 0.01, 0.97], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.3671875, -0.573486328125, 4.025390625], "student_probs": [0.024886801838874817, 0.009714928455650806, 0.9653982520103455], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.01, 0.97], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cf0bbe5c008610848cda5ba94e95f7d75e24c364601e6a81bbe2f3dc33fbc2ab:action", "state_id": "b251d2c7885cb76dcf791a537394444b5d19303ca8f13663c1c2b4977874c236", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.75, 0.09, 0.16], "teacher_probs": [0.75, 0.09, 0.16], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.771484375, -0.310546875, -0.20703125], "student_probs": [0.5831668376922607, 0.19763897359371185, 0.21919411420822144], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.75, 0.09, 0.16], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "dad14a24113020cd181c4cc2a90104186eccb9c93edb9765262f4cdce6c34249:action", "state_id": "99985a9d8a67042e29b8b86d45266e59f589bc44a7894d76db8a208cbdbe2f4e", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.62, 0.15, 0.23], "teacher_probs": [0.62, 0.15, 0.23], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.35546875, -0.24609375, -0.35546875], "student_probs": [0.4904032349586487, 0.26871880888938904, 0.24087797105312347], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.62, 0.15, 0.23], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d43460a2b7880b17dd18e9511810bb9c78d9bd866a73ac3cf6c2229647496189:action", "state_id": "fd63a11ebd12584ba7698432ac1dde8e6ee40e00d1e0c2c0ad1fbcb6143cf1d0", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.57, 0.24, 0.19], "teacher_probs": [0.57, 0.24, 0.19], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.2421875, -0.09375, -0.380859375], "student_probs": [0.44425180554389954, 0.31749245524406433, 0.23825573921203613], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.57, 0.24, 0.19], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "386a643f239db386ea7ffc452c10384fa3a58730805bd37c05ec9737a67df53d:action", "state_id": "87484dd23187c61c141b176e88e7f97ddefd1427c898aa2fe51707834220af3f", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.12, 0.39, 0.49], "teacher_probs": [0.12, 0.39, 0.49], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.140625, -0.115234375, -0.287109375], "student_probs": [0.16297073662281036, 0.4543924331665039, 0.38263678550720215], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.39, 0.49], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b307076a021e0b7368cc78a5cd7c2ee239c988241208b2a9d36a5ce3248769ae:action", "state_id": "ca2d696835bb9c010f8c3238ad5395c234023f624e9ebebca5b841deb88ff795", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.72, 0.26], "teacher_probs": [0.02, 0.72, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.6015625, 0.71484375, 0.22265625], "student_probs": [0.02202211320400238, 0.6069542169570923, 0.37102365493774414], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.72, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "69b0524f218f5126838cdb3fc028f969c6f72defc7ce70e71b0293b9bab82a81:action", "state_id": "b56a3c3f0f99b9f03c2fc62d5ed0eced70fe3b6956f5580a2faf0f711fad4808", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.72, 0.26], "teacher_probs": [0.02, 0.72, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.5078125, 0.7265625, 0.33984375], "student_probs": [0.022915907204151154, 0.581846296787262, 0.3952378034591675], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.72, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ddf8baad95a2b8a91936fd17576f6186bf73d87fe0e4da172595bad1a289a818:action", "state_id": "48fa7698d07e0585c3314bd49b456589d54fc3ed565a527bfc831c046dbdd528", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.15, 0.84], "teacher_probs": [0.01, 0.15, 0.84], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.9453125, -1.19140625, 0.97265625], "student_probs": [0.017520714551210403, 0.10121935606002808, 0.8812598586082458], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.15, 0.84], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "af5acd160dcc4ed2b1c5dd813c724f8268a3b792a70708143b2e81a5ac55e2d4:action", "state_id": "f79b05571f53cd8f81e64cdc318254aabb2b8e71b9d34ff19020b0d3e0640705", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.01, 0.93], "teacher_probs": [0.06, 0.01, 0.93], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.076171875, -2.79296875, 1.62109375], "student_probs": [0.1532546877861023, 0.010128004476428032, 0.8366173505783081], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.01, 0.93], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b0232dafe0e96716345c95bd34ea9a132aa43cd562a72a9e1f6fe04141fa7ac5:action", "state_id": "83818c80bbe90ebc785e9f7515c90bbebf511515a7a24cd112f12896e5c7fff7", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.01, 0.95], "teacher_probs": [0.04, 0.01, 0.95], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.30078125, -2.33203125, 2.181640625], "student_probs": [0.1310441493988037, 0.009418933652341366, 0.85953688621521], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.01, 0.95], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "aca04784b194317a7b0ed68a3699840fe491f76959a42b198eef623f612ca58b:action", "state_id": "71220dddb00065135044f95b4044da7212a167372cb56717273e1c51fc647793", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.0, 0.95], "teacher_probs": [0.05, 0.0, 0.95], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.619140625, -1.90234375, 2.43359375], "student_probs": [0.13854140043258667, 0.01113045308738947, 0.8503281474113464], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.0, 0.95], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "26011916b8454e5bf6c271e9b7f357a572dcc670da4d056f4e1fcb88819bea51:action", "state_id": "e60bb5ac9f0858222208c452ed080a9a51089f6d63c778f320ef752711cf62ee", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.01, 0.9], "teacher_probs": [0.09, 0.01, 0.9], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.58984375, -2.20703125, 2.169921875], "student_probs": [0.1690235435962677, 0.010310501791536808, 0.8206659555435181], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.09, 0.01, 0.9], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2cf19f0b48290a4a4d37a11f16764a71d0e1ada8b61af46618d5bbcfe0bbf7dc:action", "state_id": "7657c2fc30302c1d6c7d3d2930cb0ab4e7c88f9904a1d2dda1e0b3eba14a8a78", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.01, 0.9], "teacher_probs": [0.09, 0.01, 0.9], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.53515625, -2.2265625, 2.482421875], "student_probs": [0.12387461960315704, 0.007826779969036579, 0.8682985305786133], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.09, 0.01, 0.9], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cd877cc9d136779e53a6c834f047648ab80d90bf334c131b5877cad7bb9dc7ea:action", "state_id": "5e788ca2184144b438d32e745d55c1ff5ea7286a8bd56ed98da91b5ca2ea8451", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.0, 0.96], "teacher_probs": [0.04, 0.0, 0.96], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.45703125, -2.26171875, 3.6796875], "student_probs": [0.03822535648941994, 0.0025212352629750967, 0.9592534303665161], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.0, 0.96], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "030ecf04d9d327f5705477904b6510b60fa21881abb8336bdf464594d06ccd31:action", "state_id": "9e1923dc8dba27a37cb3e44dc8f578f488b06fb36f13f068cb57c7c127e742b1", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.97, 0.01, 0.02], "teacher_probs": [0.97, 0.01, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.65380859375, -3.00390625, -2.36328125], "student_probs": [0.7835019826889038, 0.07471463084220886, 0.1417834460735321], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.97, 0.01, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0a7c983615facd8eb6a3121e89670694182ebd3dedc7f6f93d080388ea8b73f9:action", "state_id": "ac01cd94d493203f603a6f3b13481e6e481a175634a51c142759e6f9f20928d8", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.39, 0.6, 0.01], "teacher_probs": [0.39, 0.6, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.3203125, -0.5673828125, -3.12890625], "student_probs": [0.5430722832679749, 0.4241860508918762, 0.03274167329072952], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.39, 0.6, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "052eefab01e4aa6dd99a1ad3edbe034ea536252f2776f7b73c66b96ad8a80798:action", "state_id": "ab2a1505b530300e90f407fc94b52c897e8968abb4d0682adc170a4519261e66", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.87, 0.13, 0.0], "teacher_probs": [0.87, 0.13, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.20703125, -0.476318359375, -2.984375], "student_probs": [0.6468151211738586, 0.32659175992012024, 0.02659316547214985], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.87, 0.13, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ba8a122d1e88e625798e721b5eefa6bf8844271bc7e86ea49eb0e47a8880dd1d:action", "state_id": "23df87b9a2299d92defb67bd198d73a27a2ea1171d460815d56a806903e636da", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.93, 0.05, 0.02], "teacher_probs": [0.93, 0.05, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.2578125, 0.240234375, -0.87109375], "student_probs": [0.8498033881187439, 0.1130044162273407, 0.03719219192862511], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.93, 0.05, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6fe27d247cfaacae953fe35cb77e4a5b367b95acc1035173db01a0a4cf2ecb97:action", "state_id": "a51f74f9813832aea0a7400b8bf02f6a94559c4e8bec502275bd5a7ff13c1685", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.96, 0.03, 0.01], "teacher_probs": [0.96, 0.03, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [3.216796875, 0.677734375, -0.896484375], "student_probs": [0.9129964709281921, 0.07207227498292923, 0.014931166544556618], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.96, 0.03, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7055c3f905672d6084a06b91fde1f5e0cd22161beaef1f1931713c9187d8e76c:action", "state_id": "3ae8bcd27a4041cdf4f52bfa0c29bc74970864283ac13ea2df2973b927c6b008", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.99, 0.01, 0.0], "teacher_probs": [0.99, 0.01, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [4.322265625, 1.263671875, -0.48193359375], "student_probs": [0.947733461856842, 0.04449957236647606, 0.00776692247018218], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.99, 0.01, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a98bd48ab3d7a21a77b5e1aa547bbbfbb027190aca050f1866e94376ffe0d110:action", "state_id": "b5ec76f1cd61d7d3fa8ee739a0e8d6ff93986348a742dc6a961dfd1a47572641", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.7, 0.29, 0.01], "teacher_probs": [0.7, 0.29, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.123046875, 1.34375, -1.53515625], "student_probs": [0.6736242175102234, 0.3090105950832367, 0.017365219071507454], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.7, 0.29, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "98dc09a9e716908c365f633219820fbc8b71e0ebf1d5fefd0f63dd21141f83ef:action", "state_id": "f388a5aa1cb5d7309218d9d708f8d3090d8db443d4ec24e7e3079a677647f532", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.35, 0.63, 0.02], "teacher_probs": [0.35, 0.63, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.9765625, 1.400390625, -1.53515625], "student_probs": [0.38329923152923584, 0.585604190826416, 0.0310965608805418], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.35, 0.63, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3ba5043b5565cedf7020a1e415add6b77a932543f7e193f709d1143d4d3f5f4b:action", "state_id": "31378a8dd9a308e54470d3b38779508045a284087ee9b199f68403c79d61c2bc", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.98, 0.01], "teacher_probs": [0.01, 0.98, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.40234375, 2.556640625, -1.2578125], "student_probs": [0.01832858845591545, 0.9604927897453308, 0.02117864415049553], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.98, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9796b0bafced7211e5c4a9156019bfbc348c5bba8993d930533e1e440387dabe:action", "state_id": "7e13d93d98cde3349b619e71859e3a83a699902483b3ac023e1cf99c3e2a2e99", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.0, 1.0, 0.0], "teacher_probs": [0.0, 1.0, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.849609375, 3.7734375, -0.8017578125], "student_probs": [0.009629015810787678, 0.9802699089050293, 0.010100982151925564], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.0, 1.0, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ed9a19da9d71bbdc6fd82d9c1f7245d9fa561c4263793dbf5744911cb9d72d9e:action", "state_id": "91c8c6b0b4f2bd6d8a3892d371637744911aa47df85e9284dd4bbe136f23bb1f", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.0, 1.0, 0.0], "teacher_probs": [0.0, 1.0, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.7158203125, 4.708984375, -0.7021484375], "student_probs": [0.004367178771644831, 0.9912055134773254, 0.004427296109497547], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.0, 1.0, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e1b56f1612b5ccc2813292138daa622d3d3ee5342460dc2b6dbc542f135a946a:action", "state_id": "cc02e360de53732c27019aec1d868b282177221b2abf6de48d991a2457caab3c", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.8, 0.09, 0.11], "teacher_probs": [0.8, 0.09, 0.11], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.1953125, -2.27734375, -2.0390625], "student_probs": [0.8393349647521973, 0.07080670446157455, 0.08985837548971176], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.8, 0.09, 0.11], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "752bc141c03adf7fc22fb86faf60a006df4ea652584f1ea2e25647c8c365b21a:action", "state_id": "4c873ae5119471f4c2236258bc25fcf7d9bac474df559a2a518b86e8ede297ed", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.41, 0.03, 0.56], "teacher_probs": [0.41, 0.03, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.24609375, -2.85546875, 0.16015625], "student_probs": [0.38838598132133484, 0.028577640652656555, 0.5830364227294922], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.03, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "08f94f1eae914502f623298d2894ebce1bfd68a2f45bbe10394b71449946993d:action", "state_id": "a37d9eef87792359bbf19080cfb8907098ebf524ff010851b65be4767bf7b88e", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.23, 0.71, 0.06], "teacher_probs": [0.23, 0.71, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.849609375, 0.48046875, -1.41796875], "student_probs": [0.18699301779270172, 0.7070839405059814, 0.10592294484376907], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.23, 0.71, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9d2f6b4d31e37e3d3862c3a632d824cd673b1271f52ca69909772abc517b169a:action", "state_id": "e5876264c472093d7165273617d1fa3741ebfc4b7c5cc2fde4777b84a9d4c747", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.11, 0.87, 0.02], "teacher_probs": [0.11, 0.87, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4580078125, 1.345703125, -0.84765625], "student_probs": [0.12904168665409088, 0.7835590243339539, 0.08739927411079407], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.11, 0.87, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9cdc2f9eaacc383f4a3feb5f58576d861ac5e75d2f41086cacacd8c5bc2b4090:action", "state_id": "324b3cdc0b2d6c0ff7de1730225627cfdc4cf2f105ce347da98d9ad4c3650b9a", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.91, 0.03], "teacher_probs": [0.06, 0.91, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.16015625, 1.427734375, -0.447265625], "student_probs": [0.15051522850990295, 0.7365336418151855, 0.11295109987258911], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.91, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "95eeaac76639193b6185c1a101ac8b7e6b79d15c26d8419bfdb62160798a4db5:action", "state_id": "0b43a3e7b568ef989590504e4483b7f7486cc60fd336c421458aef51fac81aa6", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.95, 0.02], "teacher_probs": [0.03, 0.95, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.0859375, 1.671875, -0.24609375], "student_probs": [0.1306890994310379, 0.7579624056816101, 0.11134851723909378], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.03, 0.95, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c9a73d89aed5f5dac64dc5c6d78bb341542292c9efb31b51756ca8cfa1dee1de:action", "state_id": "165f1ac65c7a618f82a2277791ff77d833f7fc5bac2ea34a4fb774accb3889b2", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.98, 0.01], "teacher_probs": [0.01, 0.98, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.171875, 1.78515625, -0.4921875], "student_probs": [0.11358210444450378, 0.8039661645889282, 0.08245176076889038], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.98, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fced09de237116cdb957942378b2e693f061bec97844b8294d5f8260a0f9386f:action", "state_id": "9bb6fe0ce994aee7c6b1e82e8f93896aa15cf9214423a250ad57e5c42cb1f85a", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.0, 1.0, 0.0], "teacher_probs": [0.0, 1.0, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.01953125, 4.2637939453125, 0.125], "student_probs": [0.013924553990364075, 0.9706019759178162, 0.015473402105271816], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.0, 1.0, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "13170e04480819f3ed306a08c3b80e78b6d5863fa2a7a1e69ed1cfe4dc3f3300:action", "state_id": "d2261c8d8a419c660c818bfa56c1c87bd881018f08781a94654e2a9baacac186", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.64, 0.34, 0.02], "teacher_probs": [0.64, 0.34, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.654296875, 0.873046875, -1.64453125], "student_probs": [0.42645806074142456, 0.5307356119155884, 0.04280632734298706], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.64, 0.34, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6e89e06ac16b8a08b56ac2a6844f9dd5dedd2d725b0602dc56211412f79318a4:action", "state_id": "c83978dec1c047949ee93f2ed3f60249c1ccf4d62da8997ec7226b8eead0f57a", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.47, 0.5, 0.03], "teacher_probs": [0.47, 0.5, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.189453125, 0.76953125, -1.2734375], "student_probs": [0.3313733637332916, 0.5918918251991272, 0.07673478871583939], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.47, 0.5, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4b9e7f5bf4a6e738ce1dd046166568ca747b1a8ebccfb86053a5ba6679b7e9e5:action", "state_id": "794b83fe03f933d91c411411f803a59dcae2ae1e2bf8ae4ab70938c62d22d02f", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.79, 0.14], "teacher_probs": [0.07, 0.79, 0.14], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.7890625, 0.9375, -0.17578125], "student_probs": [0.04694940149784088, 0.7173995971679688, 0.23565097153186798], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.07, 0.79, 0.14], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "51fde17a83bfcf77e86415a72b59a46b379d86102a4dbc24cac1bba8e962d442:action", "state_id": "38c38b034fc97a45de533178920f15509bc89f5ddec8dc3e0b53632492ce488c", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.49, 0.42, 0.09], "teacher_probs": [0.49, 0.42, 0.09], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.033203125, 0.732421875, -1.0078125], "student_probs": [0.29715245962142944, 0.5979242324829102, 0.1049233004450798], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.49, 0.42, 0.09], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7a0712e189d0d498ab5a5c6204a0ca960926015a4803d02bdf640e15b840e496:action", "state_id": "113a87fe6bcd02061180019d8b755543b5afd30f0fc91624f3cc1589ce3b5d7b", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.09, 0.66, 0.25], "teacher_probs": [0.09, 0.66, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.76953125, 0.625, -1.0859375], "student_probs": [0.07171522825956345, 0.7862181663513184, 0.14206671714782715], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.09, 0.66, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e91a88cb4f4c694ec595827b6f7ef53f2d42615b1b05e84c91c35e4f03b6ed45:action", "state_id": "77dfd4f1ba0febf772184e7f659dd6659e44ec0acab5f41535833ed62f754163", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.66, 0.32], "teacher_probs": [0.02, 0.66, 0.32], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.9921875, 0.296875, -0.3564453125], "student_probs": [0.023939840495586395, 0.6420117020606995, 0.3340485095977783], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.02, 0.66, 0.32], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bfdd8e9736bcf48712f22fc2defb2d061d6b0f8b123eca1f41ac370a86801809:action", "state_id": "0b79942042364aae51f382b96ae9fc70d810d0d064853732dc3259f28e627c1b", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.44, 0.55], "teacher_probs": [0.01, 0.44, 0.55], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-2.87890625, 0.2734375, 0.52734375], "student_probs": [0.018334228545427322, 0.42885270714759827, 0.5528129935264587], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.44, 0.55], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2453095742505adb3e84c9eae00404196ee17ad702da43f15731be1874ddd1cc:action", "state_id": "b1ba72e3a24aef899b37a7799a45a9986e7c8bf1af058d695fc61b0fd9b29da8", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.24, 0.1, 0.66], "teacher_probs": [0.24, 0.1, 0.66], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.7734375, -0.755859375, 1.455078125], "student_probs": [0.313105970621109, 0.06784629821777344, 0.6190477013587952], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.24, 0.1, 0.66], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "186fb29df38809d6fc44f631e16cde8b9cc0bf0433976e0e186d46deba884a69:action", "state_id": "8d2bcb5f94d8575f3e8c98630251b156c0c42eaa01c71dfb1a097121a8bae04c", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.25, 0.09, 0.66], "teacher_probs": [0.25, 0.09, 0.66], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.146484375, -0.34375, 1.830078125], "student_probs": [0.3118855953216553, 0.07027401030063629, 0.6178404092788696], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.25, 0.09, 0.66], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1defa53771a8db070c258b4b3fc7ee31319a3de380f4191d63a299de7f89b8b0:action", "state_id": "c31bacf50b348647277456f34565aae38e793e46c51a15d8edaba6cb12473a82", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.29, 0.05, 0.66], "teacher_probs": [0.29, 0.05, 0.66], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.544921875, -0.0625, 2.1015625], "student_probs": [0.3395349681377411, 0.0680440291762352, 0.5924209952354431], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.29, 0.05, 0.66], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "61e6d5322fcf54bba068c4e00a3b9be8a0de652c45ce6b20635833f6e8d556af:action", "state_id": "f51ad215cc27a303500856a9905e4b970dc945e99e19db856f877e027f6da6cd", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.41, 0.06, 0.53], "teacher_probs": [0.41, 0.06, 0.53], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.609375, 0.0859375, 1.71875], "student_probs": [0.4285331070423126, 0.093403659760952, 0.4780632257461548], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.06, 0.53], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "207a6fa363455303122e6d924427cf8730f77c972634dadfd2a332bbc16d401c:action", "state_id": "91ed3857c61fdbbb02294a64b3dd427058d36acdd9ecbef89737b9d18b3cdde4", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.35, 0.05, 0.6], "teacher_probs": [0.35, 0.05, 0.6], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.064453125, -0.125, 1.462890625], "student_probs": [0.6024340391159058, 0.0674593448638916, 0.33010661602020264], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.35, 0.05, 0.6], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e0efaee445aea1a1b9f41ddda426fb811901b9a98ead112b2269626eeed9798d:action", "state_id": "de8fefa384e7565823c0862f708268101bf1328db32199e3473287ca9a9ee35d", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.92, 0.02, 0.06], "teacher_probs": [0.92, 0.02, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [3.6875, -0.5955810546875, 0.634765625], "student_probs": [0.9424806833267212, 0.013006307184696198, 0.04451298713684082], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.92, 0.02, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8cbd3869fbcfc8ea995d0f0e1402c54050cbb907d120fe09bee716e572806d7e:action", "state_id": "3f4f417a8c4c1d7114d5bbb0b1857ee30a7e07b7b3927c278f690f4f9775717c", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.68, 0.16, 0.16], "teacher_probs": [0.68, 0.16, 0.16], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.845703125, -2.140625, -2.1484375], "student_probs": [0.6469532251358032, 0.1772129386663437, 0.17583386600017548], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.68, 0.16, 0.16], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fd0bf229ef54498d4b6029e6a3334b84e487569d954a1a585785e49e91773a8d:action", "state_id": "316dc4048105492151c0e7006df30f1e67dabe745772dc9cf24b5e07ad34ada6", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.58, 0.03, 0.39], "teacher_probs": [0.58, 0.03, 0.39], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.15625, -2.83984375, -0.544921875], "student_probs": [0.5726478695869446, 0.03912169858813286, 0.38823047280311584], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.58, 0.03, 0.39], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "475107659620f18151a99a49ee037d11758efdb946534c2d656e6873b8f53b85:action", "state_id": "39fa94c3160fcbd679f821887756c746cb45811534befcc381355c835174e2fa", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.5, 0.03, 0.47], "teacher_probs": [0.5, 0.03, 0.47], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.3115234375, -2.87109375, -0.45361328125], "student_probs": [0.5141701698303223, 0.039764873683452606, 0.44606491923332214], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.5, 0.03, 0.47], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3137b10492357d43991b2b1c1bc398a8222c2f6b70ff0abf057bf93fc1077274:action", "state_id": "58d8dcba5deb6a675617fb5fe12f04c23577f74811ec6a0487977ffa5962949f", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.52, 0.02, 0.46], "teacher_probs": [0.52, 0.02, 0.46], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4959716796875, -2.79296875, -0.37109375], "student_probs": [0.4477136433124542, 0.04502224549651146, 0.5072640776634216], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.52, 0.02, 0.46], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b5393d11c6da98b00ea679643ed635435a71f7a5325364c2e67849e4e290ebc0:action", "state_id": "a69bd6ba236bd272ee28e086983be1c13f983954fafe5bda942c95eb0f58f197", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.41, 0.03, 0.56], "teacher_probs": [0.41, 0.03, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.537841796875, -2.76953125, -0.564697265625], "student_probs": [0.4805731475353241, 0.05158804729580879, 0.4678388833999634], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.41, 0.03, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a3d321115b60f0814de3e47a07c495041d34d02ce3ced6e181e51179d0f118a2:action", "state_id": "aad560cc7c6499621ca1311003b7e8ca19ad9dabe923d5d2d8426d6b1a9d1317", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.58, 0.38], "teacher_probs": [0.04, 0.58, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.4921875, 1.01171875, -0.044921875], "student_probs": [0.05720284581184387, 0.6996008157730103, 0.24319638311862946], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.58, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b184b421663a85c3b5e0a3f4e99bc88eb69d2dd35e0d64f6e74d1fc808705999:action", "state_id": "c32404c6b71219a719a2d20e54419849a45bf16e4574dd2f4518841d309579ee", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.57, 0.38], "teacher_probs": [0.05, 0.57, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.078125, 1.20703125, 0.501953125], "student_probs": [0.06376511603593826, 0.6266339421272278, 0.3096010088920593], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.57, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "024027960eea6a063565b01fe9dda747e31d4bb09355ef49ca5bf5006a8c8d3a:action", "state_id": "365c179c33f51874ed92edfc49ef2f8261a77b63811e2b173c1716ccbbdb0ee8", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.38, 0.61], "teacher_probs": [0.01, 0.38, 0.61], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.2578125, 1.158203125, 1.68359375], "student_probs": [0.03210932016372681, 0.3596610128879547, 0.6082296967506409], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.38, 0.61], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0ae0b307c6e701c46c8d60a94454cbd3b3ba4ebcb4ed5e54cc2b16b776e51266:action", "state_id": "aea171a084b983c5a4a22c93cf0e14b4f97e19c70c2f59977fcfe067f6a4ff04", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.04, 0.92], "teacher_probs": [0.04, 0.04, 0.92], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.05078125, -0.16015625, 2.7529296875], "student_probs": [0.05980303883552551, 0.04842997342348099, 0.8917670249938965], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.04, 0.04, 0.92], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "dd513f7f7d9f62b79e7968dae97471b28dbecc549011633ad9ad71453435f3d9:action", "state_id": "4f5b5ce0109c26f8e2d304fb40f3aac64a68340e19eabbf83c780b1b48b3a4e9", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.03, 0.92], "teacher_probs": [0.05, 0.03, 0.92], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.2890625, -0.06640625, 2.703125], "student_probs": [0.07763897627592087, 0.05441287159919739, 0.8679481744766235], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.05, 0.03, 0.92], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "bbcf5e062fb9a0d7aec7a0685631c30d8c90bcfbacf9fb8543dbd4937fb12e0f:action", "state_id": "ff366f7a58cca8d10ac15ce18e46fe3d09411ac07fffaad4e0c9fff9ea12b7ef", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.0, 0.99], "teacher_probs": [0.01, 0.0, 0.99], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.60546875, -0.015625, 3.762939453125], "student_probs": [0.03992269188165665, 0.021452713757753372, 0.9386245608329773], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.0, 0.99], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ae9c59b0f1d6562dceade71719564d6154280b1137fe7e0ab6af9d269ce4e351:action", "state_id": "663d365da04c525b2e046ffa7ade694592c3c37986f0b6dd48fb84b99792c2d2", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.78, 0.06, 0.16], "teacher_probs": [0.78, 0.06, 0.16], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.904296875, -0.608612060546875, 0.3515625], "student_probs": [0.7735743522644043, 0.06268441677093506, 0.16374124586582184], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.78, 0.06, 0.16], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "262ead12904342099e506cf10db5f7ebe4a1abde64e3cacea960785b5c1c6131:action", "state_id": "c3a74d710945b6330fa6bb7b6d3b293792e0f90520f764f162192f6f928075ac", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.1, 0.87, 0.03], "teacher_probs": [0.1, 0.87, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.6029052734375, 0.1640625, -1.060546875], "student_probs": [0.2641308903694153, 0.5687338709831238, 0.16713522374629974], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.1, 0.87, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8912c9f2b7e1b87e67fd802dae3a220fb63862ddd45c50bfbb19f8f51d53ef82:action", "state_id": "299afa369bc48213d303163c8dd47ac1361afc6e2eb37072221c4846271a86b8", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.12, 0.86, 0.02], "teacher_probs": [0.12, 0.86, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8056640625, -0.04296875, -0.796875], "student_probs": [0.2407970428466797, 0.5162802338600159, 0.24292273819446564], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.12, 0.86, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "387ac6e9bc63ab2bdb01df5ad4964b9a88918a5c21632d6edd60b08b5c4f4273:action", "state_id": "80d3db80e4c9d35a74a600e5e6ba519f6c77d7672303e5d60e23482bce4a8d06", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.3, 0.63, 0.07], "teacher_probs": [0.3, 0.63, 0.07], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.494140625, -0.478515625, -1.4765625], "student_probs": [0.4183836281299591, 0.42497220635414124, 0.15664418041706085], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.3, 0.63, 0.07], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6fd7853e414d025d1ada4b27f8b694247274ae30ca5025232818b902ccede411:action", "state_id": "6d81a23831438030ca55ebadf063a52f4446d64d540c9625679a5befe2173e61", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.59, 0.3, 0.11], "teacher_probs": [0.59, 0.3, 0.11], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.865234375, -2.34765625, -1.7109375], "student_probs": [0.603739857673645, 0.13710150122642517, 0.2591586112976074], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.59, 0.3, 0.11], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f65cd7e5a98cfe662e471c66418383c5c5f66f7cf3d67d1ff5b7de47c933c3f9:action", "state_id": "51fe585f5a8fedb509f991f5ee24aa59c8379044a7739768d00a3a7ab4547fc9", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.6, 0.02, 0.38], "teacher_probs": [0.6, 0.02, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.5908203125, -3.01953125, -0.4979248046875], "student_probs": [0.4575617015361786, 0.04033424332737923, 0.5021039843559265], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.6, 0.02, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "240ff3a3b80d47f4cd9496394304dc055dcb7238b6580672673eafccdad08b17:action", "state_id": "aba4cb2f427aedd44202ca2a5f6dadcc6ced88760c17aee001c9c63ec8f9b151", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.44, 0.03, 0.53], "teacher_probs": [0.44, 0.03, 0.53], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.751953125, -3.1484375, -0.58251953125], "student_probs": [0.439430832862854, 0.040004659444093704, 0.5205645561218262], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.44, 0.03, 0.53], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "41a2d4cd37a24dbd5da8599293ce5b3ed08479e702fb52745ab8f10ae605b9a3:action", "state_id": "fe1c838c6986c8e4490a6b1b3b32a65f37db98547e2721a2fd359f22cf759c18", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.73, 0.21], "teacher_probs": [0.06, 0.73, 0.21], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.28125, 0.31640625, -0.3076171875], "student_probs": [0.11642823368310928, 0.5753228068351746, 0.30824899673461914], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.06, 0.73, 0.21], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9bcb489ef3c301c60c7b5fb42312b6a4a0028ca0961e405f2fc55f67731da27e:action", "state_id": "86f675dbf942243b918bcc48b2ccd06b16f20ef808e34b7dc89ebd976ea68d61", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.62, 0.35], "teacher_probs": [0.03, 0.62, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.859375, 0.849609375, 0.42578125], "student_probs": [0.09863312542438507, 0.5447851419448853, 0.3565816879272461], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.03, 0.62, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0b490659f420558524f3e411f066293aba8e0bdd8535e881c5d0d91a20807ea0:action", "state_id": "01c56ffa73f94acd9fe8ffa23f261b1009af80c52401ccac8fae2bd1ce15f901", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.64, 0.35], "teacher_probs": [0.01, 0.64, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.6728515625, 1.16015625, 0.84765625], "student_probs": [0.08455078303813934, 0.528667688369751, 0.38678157329559326], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.64, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cb035f034cdcec9ea2ccf4bc04626f6f41c59aada2ad9675179685f94de77382:action", "state_id": "fa4bf862e18cb3a9f331f31ecfd310f26f36508d30706c967968ca1b9ed0df9b", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.63, 0.36], "teacher_probs": [0.01, 0.63, 0.36], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.666259765625, 1.29296875, 0.93359375], "student_probs": [0.07665091007947922, 0.5437502264976501, 0.37959882616996765], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.63, 0.36], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7318770f1f1a5bf118c2b61077e5596235eed47b27019bcda39b7541fe19ed49:action", "state_id": "e3ac539f9348ee7c89d8a6a728ac525886724ebeafcc9bef99c12cce9b17d066", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.57, 0.42], "teacher_probs": [0.01, 0.57, 0.42], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.869140625, 1.396484375, 1.19140625], "student_probs": [0.05409087613224983, 0.5212816596031189, 0.42462751269340515], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.57, 0.42], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "895355c4e340396ae9f1f14b74578ee2672a8b639dcbeb801e1428cd26c56bed:action", "state_id": "19aef764997fe53408cec05b136ed3528b0643c08c756f084beed7d5efb3671c", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.29, 0.7], "teacher_probs": [0.01, 0.29, 0.7], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.1015625, 0.76953125, 1.958984375], "student_probs": [0.03468053415417671, 0.2252638339996338, 0.7400556802749634], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.29, 0.7], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9da20b0ce9bf1c0c18f26a6690fc1e90a7f1d1c0f44c5238c68c4047586ba3dc:action", "state_id": "9558108649039c21596ac269e03a40576556b886bf20ab912f23615bea6b72b6", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.0, 0.99], "teacher_probs": [0.01, 0.0, 0.99], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.21875, -0.34765625, 4.38037109375], "student_probs": [0.01521073654294014, 0.008633026853203773, 0.9761561751365662], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.0, 0.99], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0fb27b063842a515d0911dcb24c7c9179d1472db40e82f50a811c7547f7d3621:action", "state_id": "ed5df0dbc23f20de2eaeb1a81b2071812553ab086f5ebd1d0edc957c09af22b7", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.36, 0.08, 0.56], "teacher_probs": [0.36, 0.08, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.85546875, -0.8125, 1.01171875], "student_probs": [0.4241335093975067, 0.0800042375922203, 0.49586233496665955], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.36, 0.08, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b86ca048070b19568729b7abe92b68199e9691501d8e66b4e16c6000817a45c2:action", "state_id": "3cfccf5a6fd2a7a3f99d6aa9a9b651474a5374428db651cd9fa6b61e0343e364", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.53, 0.09, 0.38], "teacher_probs": [0.53, 0.09, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.90625, -0.990234375, 0.34765625], "student_probs": [0.5806835889816284, 0.08715792000293732, 0.33215850591659546], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.53, 0.09, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4469577474c8fd7841924f413f60cca2957db01a57e50b58a783cc8de7a8713d:action", "state_id": "ae07e05514403331f9b2dfb38a850c450fa9be9a876efaa0ac448eea4b9b6706", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.08, 0.86, 0.06], "teacher_probs": [0.08, 0.86, 0.06], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.675537109375, 1.466796875, -0.8837890625], "student_probs": [0.09679322689771652, 0.8246103525161743, 0.078596331179142], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.08, 0.86, 0.06], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c8f192cfba32b3feae3b0ca85c1b6a9fa34b89554716b1b499d74e5df52e230e:action", "state_id": "b47258a91b184d1b0e2a8b1cf49676b3802afcb19f27bef581a8ea212b89c77a", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.95, 0.02], "teacher_probs": [0.03, 0.95, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.5970458984375, 1.623046875, -0.8369140625], "student_probs": [0.09095112979412079, 0.8374947905540466, 0.07155412435531616], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.03, 0.95, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6a30b39d732b3d37b27e658ff3968e0fb3402e2f39ca9becd547cbf875e51514:action", "state_id": "9946e0b94f2733595479124ffc731dd36e456c7f03dbc605dacd3ca5b6b39d7f", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.98, 0.01], "teacher_probs": [0.01, 0.98, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.203125, 3.76171875, -0.56787109375], "student_probs": [0.027340082451701164, 0.9600136876106262, 0.012646211311221123], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "a3feb71a222fe06e13ebefbe89dcd6bce22cddf8bf5f9e02c02555de059cb2c5", "training_target": [0.01, 0.98, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1346c5a81ffe10f8d6719807fd11d5868004a51b90df6616a19054cf185f8ae6:action", "state_id": "b08104dc8337d3bd0549b311bb017f764c37c7b61474db5162436761276d7be7", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.16, 0.16, 0.68], "teacher_probs": [0.16, 0.16, 0.68], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.50390625, -1.1328125, -0.875], "student_probs": [0.23122043907642365, 0.33511218428611755, 0.4336673617362976], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.16, 0.16, 0.68], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a27fc8354c730d9572c15f9d9b9bb5cc4048ab680d9caf88257e7bf03e2a3771:action", "state_id": "652182a0fe47b799a15e84d2f1046d62537ac696b7554cea4992a6f9444b5891", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.31, 0.69], "teacher_probs": [0.31, 0.69], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.00390625, -0.966796875], "student_probs": [0.49072369933128357, 0.509276270866394], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.31, 0.69], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2377055066b1aa5c70107a7eac3578a5c3ff674f29cb66fe6f9480ab4599bf73:action", "state_id": "1996f099f738e43082ea1d9ad2a4ea7abc0b0ec3f4f9e8270df1717429de41b8", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.62, 0.25, 0.13], "teacher_probs": [0.62, 0.25, 0.13], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.97265625, -0.798828125, -1.001953125], "student_probs": [0.31635767221450806, 0.37641850113868713, 0.3072238266468048], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.62, 0.25, 0.13], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f258a8bae34c20bb7da8bf571dcd6d484dc65324a36738bf5a3d786b967b9e16:action", "state_id": "a086dca806ca4d9560971b1df2d36823825b3fa583bde321ec1a8c3b06d1cc5b", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.51, 0.31, 0.18], "teacher_probs": [0.51, 0.31, 0.18], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.7412109375, -0.751953125, -0.9609375], "student_probs": [0.3581593334674835, 0.35433250665664673, 0.28750818967819214], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.51, 0.31, 0.18], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f0d5e583a7e01de342294e426d66b6a211ae60e6240b9e7f8c6eb84fd7995902:action", "state_id": "270f70521296f6b9ec921eae3e679d98dbee25a1715bfeaad2f3a12a78178648", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.44, 0.4, 0.16], "teacher_probs": [0.44, 0.4, 0.16], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.931640625, -0.79296875, -0.9296875], "student_probs": [0.3173895478248596, 0.3646003305912018, 0.3180100619792938], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.44, 0.4, 0.16], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c2fc714986583e02983c654186331652d13d3b33e6bd8d87dcfd32e03d771d05:action", "state_id": "30e43a1cebb7862d71b29c144c5f4a8d16ac3dc4e51295a195b7723707270036", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.42, 0.43, 0.15], "teacher_probs": [0.42, 0.43, 0.15], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.76416015625, -0.7626953125, -0.8994140625], "student_probs": [0.3478309214115143, 0.34834080934524536, 0.30382826924324036], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.42, 0.43, 0.15], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "dbc2214a903b7ff35ff487793d2f0408aced2c2b47cad4a64555bce8f6cfc568:action", "state_id": "141b3b5873fedfb0dedf42f80acba42e205a487db2667cf4208069ce3f44513d", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.42, 0.4, 0.18], "teacher_probs": [0.42, 0.4, 0.18], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8603515625, -0.865234375, -0.88671875], "student_probs": [0.336801677942276, 0.33516114950180054, 0.32803723216056824], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.42, 0.4, 0.18], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "65b9edd6267d6799c60ec40470298c5264c09f6db193316a2a01c4ffeccfc4eb:action", "state_id": "bec68620c18e6da9e03ba5426bd0d232cd3ed373ed26cc899f4982bcc512492f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.27, 0.73], "teacher_probs": [0.27, 0.73], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.017578125, -0.74072265625], "student_probs": [0.43122485280036926, 0.5687751173973083], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.27, 0.73], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8fb803f1a28560e6310bcc85a0348850c16f85b6bcf1e64c2ea15b8a1241cdc4:action", "state_id": "16217c51f2768bc7d8b21e94c0689143430f98a9c8c6bf119b9cb16a9308e389", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.22, 0.78], "teacher_probs": [0.22, 0.78], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.25, -0.66943359375], "student_probs": [0.35880228877067566, 0.6411977410316467], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.22, 0.78], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b72f80aad47004a59d9514356e72482328948cd9589fc636a62c65794db7eab4:action", "state_id": "768aaaa9c10e0408dcdc0f4d49fb868777a9122bf350250bfd1b36a7d9339a42", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.27, 0.73], "teacher_probs": [0.27, 0.73], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.033203125, 0.021484375], "student_probs": [0.25832599401474, 0.74167400598526], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.27, 0.73], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "28e01ac1a98150f2be38443978daa11f1d029aaa5c752f21177c7d3ac23f276e:action", "state_id": "2db7596a76bf15738910d2f9259086cea028c83f2929c9f7874ef899c778a482", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.51, 0.49], "teacher_probs": [0.51, 0.49], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.04296875, -0.8583984375], "student_probs": [0.7112303972244263, 0.2887696325778961], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.51, 0.49], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "805e40cb1f1a689bc440450bdad8ed4061acdfd51aef1390d12e31ce1d6a3358:action", "state_id": "20a16f2f40ad6e41479c693d436c55e6b6bde386552d41fb834396e797bf3e49", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.31, 0.29, 0.4], "teacher_probs": [0.31, 0.29, 0.4], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.20703125, 0.046875, 0.4453125], "student_probs": [0.32040226459503174, 0.27298614382743835, 0.4066116511821747], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.31, 0.29, 0.4], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b90e02ee90e703ab42c8da11427ee7194c24257d4fd6afdd0744058acf78e660:action", "state_id": "78c94b401fda97473d8342955f43eb555186cde8f3d9a3f10edb8a3dad6eb370", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.66, 0.34], "teacher_probs": [0.66, 0.34], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.0859375, -0.0859375], "student_probs": [0.5, 0.5], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.66, 0.34], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f161a2fe05ba756a3d05bbe8944f2685b97a2211ef81fa5a3b0b304a2b421322:action", "state_id": "acb06af22cc6f204d8b28e79913fbb273a24631d0f103ac9f855cc2e82636c88", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.19, 0.81], "teacher_probs": [0.19, 0.81], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.986328125, -0.994140625], "student_probs": [0.5019530653953552, 0.4980468451976776], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.19, 0.81], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1efd34fe73fa5b835524f3336fabf56b44b461d6e5d114bad961cc049eb2cf5d:action", "state_id": "e771c418b3d9340f14c60d0d169f43504206bc9a3a83dcd56655d2de05688e03", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.3, 0.61], "teacher_probs": [0.09, 0.3, 0.61], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.359375, -1.27734375, -0.779296875], "student_probs": [0.25828662514686584, 0.2803674638271332, 0.4613458812236786], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.09, 0.3, 0.61], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1231a4d38876dead54cc07bc682ba6df835d0233ebfe1e01ac9c03ee31eef4c4:action", "state_id": "472c7210935ddb3cf5389cf60a2fe9cbcf8eb21614866480294c56551f5070d4", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.44, 0.15, 0.41], "teacher_probs": [0.44, 0.15, 0.41], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.72802734375, -1.21484375, -0.80859375], "student_probs": [0.3941393792629242, 0.2422301173210144, 0.3636305034160614], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.44, 0.15, 0.41], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a601fae54e0cce7ec6770489bfd83be496ff2416422a2026762c0d90eb495d68:action", "state_id": "41b8f0a4edb250aed608bab431c1dbd5f99438ea3a432c6873d48b7c5e295edb", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.34, 0.66], "teacher_probs": [0.34, 0.66], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.9130859375, -0.05859375], "student_probs": [0.29849135875701904, 0.701508641242981], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.34, 0.66], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ff4e17ab4ea1776f3df17e10aaf6245cefc9fb848a56b8a1994afa6716b58f42:action", "state_id": "a7fa3d0accf76fac9f9bb9762a62af0611425898525cccb9c00b923ffa47cfc7", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.35, 0.13, 0.52], "teacher_probs": [0.35, 0.13, 0.52], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.2578125, -0.8623046875, -0.021484375], "student_probs": [0.3554997444152832, 0.194227933883667, 0.4502723515033722], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.35, 0.13, 0.52], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5896880248f8edecf4c59e536d5694759752f67f23f08d601598bdd8d8227c52:action", "state_id": "5d674e767a0ef71dcb5120b1a14fc50355535773d4ff988f66064fc46180f1fd", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "west"], "gold_index": null, "teacher_raw_probs": [0.74, 0.26], "teacher_probs": [0.74, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.08203125, -0.64306640625], "student_probs": [0.6737285852432251, 0.3262714445590973], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.74, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8035f89086bdab9ab409f3bffa51ae07aa201218f6b5d5e3f0fe21aba58a44a4:action", "state_id": "7f5fbfdca47f406ba9186064210c9f132edb40a2e8ee5faaa87489a6620acfdb", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.24, 0.49, 0.27], "teacher_probs": [0.24, 0.49, 0.27], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.0078125, 0.015625, -0.31640625], "student_probs": [0.3625561594963074, 0.37115392088890076, 0.26628991961479187], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.24, 0.49, 0.27], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2dde1eb82294b33a9568fe3a2b824c7d2abdc2ea82c32b79cc2fe49b7b1fabb8:action", "state_id": "b22c8af2d73618d5f5189faa64c09b84cd30bb694a5485eddabdbc12db92bbaa", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.45, 0.55], "teacher_probs": [0.45, 0.55], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.730224609375, -0.361328125], "student_probs": [0.4088076949119568, 0.5911923050880432], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.45, 0.55], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c6c8381523ffbe7eb4f88dca71466d7948076b17d32a2fc5ba3d6d0c8ed78472:action", "state_id": "906705ca99cb94d29e46df3a8c14dbb44b762a84dca447d4815e341c388d42ff", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.42, 0.14, 0.44], "teacher_probs": [0.42, 0.14, 0.44], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.83984375, 0.08984375, 0.822265625], "student_probs": [0.4073415994644165, 0.1924145370721817, 0.400243878364563], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.42, 0.14, 0.44], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b1f25240706e0fe9d8982b72522ccd23e58b42dfed33e8bc70627af7beec78ef:action", "state_id": "f4c9e1de98bd789f1fc882b905fcb00a4d84e604f210990c668d36c71d93d0b2", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.32, 0.03, 0.56], "teacher_probs": [0.09, 0.32, 0.03, 0.56], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.033203125, -0.482421875, -0.61920166015625, -0.6173095703125], "student_probs": [0.1735149621963501, 0.3009803891181946, 0.2625037431716919, 0.2630009055137634], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.09, 0.32, 0.03, 0.56], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "43bc5f8d55fb6a938dd9ae7751bb742bff070263fdfb1d6d085a796b50498158:action", "state_id": "bd6e5b528f4d6e5f98d14755bfa49bef429353a2ced01ddfb8a56de92ebed5cf", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.47000000000000003, 0.2, 0.33], "teacher_probs": [0.47000000000000003, 0.2, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.7998046875, -0.97265625, -1.193359375], "student_probs": [0.3974694311618805, 0.33437609672546387, 0.268154501914978], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.47000000000000003, 0.2, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "49a0d0f8032d21674bb9d39635fc15700994f2cda15b0efb3e8029c39bf50a38:action", "state_id": "227d8caa9010b0ed7da1ca094e9cf6b075b0fe1f4f8846f81708b0183b0964dc", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.25, 0.42, 0.33], "teacher_probs": [0.25, 0.42, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4150390625, -0.4296875, -0.5986328125], "student_probs": [0.35489532351493835, 0.3497345447540283, 0.29537010192871094], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.25, 0.42, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "238a14bc460199ac976345c727f67b37a4009504bb6eb800e97a48074996acaf:action", "state_id": "ea80b6b3849f99509dcc684bcf35b553a9d6cae1c1a731ecd8781f8cd1ee2e4f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.24, 0.76], "teacher_probs": [0.24, 0.76], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.7255859375, -1.0234375], "student_probs": [0.5739172697067261, 0.4260827898979187], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.24, 0.76], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d5c9b8c87414f00d05a07a8038a5297782acdcb058fdbfd302994d8ae77de82d:action", "state_id": "2c04c39ef6881d71964fb369e537805e064aeca16aa5f820292accca01810122", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.35, 0.65], "teacher_probs": [0.35, 0.65], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.6923828125, -0.447265625], "student_probs": [0.4390256702899933, 0.5609742999076843], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.35, 0.65], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "85739eb050c2fffe309a67df8a429c6bb2ebfe4fd03796c3b38de9656c2a4688:action", "state_id": "2a7f49a832b06cf11218cfb3db78b6657f3ddb78465b8a7a7c291d3f9066631a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.36, 0.17, 0.47], "teacher_probs": [0.36, 0.17, 0.47], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.513671875, -0.6142578125, -0.103515625], "student_probs": [0.29313817620277405, 0.2650870382785797, 0.441774845123291], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.36, 0.17, 0.47], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cb763aaf45790c6f6ca32f2764db6c109d08eea1720177bd7d8eea83f3243573:action", "state_id": "e369754f2b9cbbbcd58d05f07e8892d9219078bde48db2f12ab79e0ab9add3ab", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.14, 0.34, 0.52], "teacher_probs": [0.14, 0.34, 0.52], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.11328125, -0.740234375, -0.69921875], "student_probs": [0.252200186252594, 0.36623311042785645, 0.38156670331954956], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.14, 0.34, 0.52], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2d9758c9d6e7cc3bf5011ad7f37a1d8670e36dc75b1b4b3d469ce373120ae3ae:action", "state_id": "af0a3a5fb02f8606d57b4b89a2fce4480b5bcd9f46fa18dc5e6d67ad01e940d0", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.24, 0.76], "teacher_probs": [0.24, 0.76], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.8447265625, -0.8515625], "student_probs": [0.501708984375, 0.498291015625], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.24, 0.76], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "31a89457ce3d9a4bac3fe78ca8d7902396212eb0b32860268da2f0a09f64ce33:action", "state_id": "825e84d113cdeb4d9933ed19185fc9716b97412df7b845024c47d122afc99a3f", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.4, 0.51], "teacher_probs": [0.09, 0.4, 0.51], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.826171875, 0.0078125, -0.3046875], "student_probs": [0.200521320104599, 0.46169522404670715, 0.33778345584869385], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.09, 0.4, 0.51], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6d29676f85404e4e3ab5611c31ce7f6919d16619e161bd19dec4236413b1ab89:action", "state_id": "fc6a1750182d79ae1776dd68e1472c8178a87de094d417151960b12e94cdb6b8", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.36, 0.29, 0.35], "teacher_probs": [0.36, 0.29, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.154296875, -0.431640625, 0.1328125], "student_probs": [0.32358649373054504, 0.2452118843793869, 0.43120163679122925], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.36, 0.29, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e3ff707e6d27562daddbe4e4b680f0e012c5c63e2d46fc983f80928180541508:action", "state_id": "3796655a65a34a4057679e710d371453f75ea15f6e522e8b33ea0f7ac3b0dc70", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.37, 0.25, 0.38], "teacher_probs": [0.37, 0.25, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.177734375, -0.388671875, 0.16796875], "student_probs": [0.3102884292602539, 0.25127923488616943, 0.4384323060512543], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.37, 0.25, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3d7edbaf97ffc363f38cd2337693b4d5666104cafc394e68360536d7b225d901:action", "state_id": "d7725537f43d75926c81fbf5edb18871ab8e27178b75ee86ee549e8354c759c2", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.26, 0.24, 0.5], "teacher_probs": [0.26, 0.24, 0.5], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.10546875, -1.015625, -0.435546875], "student_probs": [0.24703018367290497, 0.2702518403530121, 0.482717901468277], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.26, 0.24, 0.5], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "30778036b73f372f2810864beaeaabcf1f61072c9b5054320588ebc59f2af908:action", "state_id": "e59a2eea2857fcb6a6b6fce0856cc618b0e5ef3e3834d48cf08b909f81587abd", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.2, 0.32, 0.48], "teacher_probs": [0.2, 0.32, 0.48], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.125, 0.25, 0.708984375], "student_probs": [0.21019595861434937, 0.3058333098888397, 0.4839708209037781], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.2, 0.32, 0.48], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "3d724a35a32696beefb042c1d80c26b3dfdc04b74714e9ff587cfa371fe9a13c:action", "state_id": "5b8b6a715f41ef4d385f78b7dd2cad25032a2feac7026f0ec399b6fc23a9eaf7", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.45, 0.55], "teacher_probs": [0.45, 0.55], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.09765625, 0.34765625], "student_probs": [0.4378235340118408, 0.562176525592804], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.45, 0.55], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1306b4decf07315794546797c07e63516825364e6bbf5811aa38b88f59e8941b:action", "state_id": "759753750a5ce88649bd43c8d0cd357f42966242465c7f39ff11a090996b5f3a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.16, 0.58, 0.26], "teacher_probs": [0.16, 0.58, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.6171875, 0.28125, 0.501953125], "student_probs": [0.15342167019844055, 0.37676724791526794, 0.4698111116886139], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.16, 0.58, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8aec15b1843571e22c7f67db6bf05cbdd4659c7d33187c86245c8a4f5cbd4ebb:action", "state_id": "17e6730c40a03ad8f52b8e0af0ba730346eddf5695a2fd912986f161073018bc", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.2, 0.8], "teacher_probs": [0.2, 0.8], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.513671875, 0.2734375], "student_probs": [0.3127896785736084, 0.6872103214263916], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.2, 0.8], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c18d03ca50dde81c74889c819a6cd402028514da3c7711c2628d8f3b5b727171:action", "state_id": "ef39899ac5293db504e4c40a34ea68d9354720c3710d7bc4246d07f9cbb1ee70", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.31, 0.69], "teacher_probs": [0.31, 0.69], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.6910400390625, -0.56591796875], "student_probs": [0.468760222196579, 0.5312397480010986], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.31, 0.69], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9db9663f6d672c694dc42faa419981207ba60507323bed37091a744eb5af129a:action", "state_id": "f07a9c168a2a5634e78946e97d5ab81909bf0b88ab0539a113619644f94b9e20", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.39, 0.48, 0.13], "teacher_probs": [0.39, 0.48, 0.13], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.166015625, -0.208984375, -0.80078125], "student_probs": [0.4836552143096924, 0.3324110507965088, 0.18393370509147644], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.39, 0.48, 0.13], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cbcb94a7ca33b53b28344d1e4d258937fc5fcb1fe08de9b2c7a3d95f4b048a9c:action", "state_id": "e38a9978b92a81f7b518d3a8d5b11ff2776c0035c24f27a5c937ad8f2ddab76d", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.4, 0.52, 0.08], "teacher_probs": [0.4, 0.52, 0.08], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.125, -0.140625, -0.79052734375], "student_probs": [0.4614606499671936, 0.35381415486335754, 0.18472522497177124], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.4, 0.52, 0.08], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9ca534630fb997d1616866b3809bd26faa0135f85e39611c6ab29e7edc4d2d3e:action", "state_id": "190d0be73b50c991f941a327ef1bf2489901ba6bea6332b9e0fa9b1fb4502ca2", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.65, 0.21, 0.14], "teacher_probs": [0.65, 0.21, 0.14], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.572021484375, -1.171875, -1.076171875], "student_probs": [0.4644874334335327, 0.25495344400405884, 0.2805590331554413], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.65, 0.21, 0.14], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "80e84d41c1a32105aa47e0f62fd948e56204de65423d89b8f39285c87b3991ea:action", "state_id": "f9f6ee6d8d424804cc4538cb80e1259f152a6614292895bbcf20341844bf379a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["south", "west"], "gold_index": null, "teacher_raw_probs": [0.33, 0.67], "teacher_probs": [0.33, 0.67], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.076171875, -0.84765625], "student_probs": [0.4431183934211731, 0.5568816065788269], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.33, 0.67], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e0654f71568fedc5250160f2a1b9990dc19c5bed5bf800635f49bc3844f971c9:action", "state_id": "5bdbd7dd2e448c892fac74c4f1d00eba527a555b3eeb0f9fefc2377454719be1", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.28, 0.23, 0.49], "teacher_probs": [0.28, 0.23, 0.49], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.3359375, -0.60302734375, -0.494140625], "student_probs": [0.38178423047065735, 0.2922956347465515, 0.3259201645851135], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.28, 0.23, 0.49], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0a8efc4e8d994333a2dac3ad32a84afa797e122479b6e605529e2a4f1c942d09:action", "state_id": "bd61343f7c6a6e8a67991d37c4ed1d30f152063a5163dd7cf6e4c667da6c8cb5", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.5, 0.5], "teacher_probs": [0.5, 0.5], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.29296875, -0.380859375], "student_probs": [0.5219585299491882, 0.47804147005081177], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.5, 0.5], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "934ec509ed643a1b07718330b1d3e55dcb82d6d61e8d5285f2840f984bda8936:action", "state_id": "8c9a45be6351340ec851c2f732cecf7b96cfd73a8fa5221d290e6439917f4040", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.3, 0.38, 0.32], "teacher_probs": [0.3, 0.38, 0.32], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.271484375, -0.173828125, 0.47265625], "student_probs": [0.237686887383461, 0.2620696723461151, 0.5002434253692627], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.3, 0.38, 0.32], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0ca04652855bf239a8eb0e592638209ee84e75f91d000dc59708e5d132d7c1cb:action", "state_id": "98e69fe92b50b25047139139b8c283e4760a02af30077cd32c56fc9f05d3ff51", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.25, 0.07, 0.68], "teacher_probs": [0.25, 0.07, 0.68], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.865234375, -0.974609375, -0.10546875], "student_probs": [0.24788251519203186, 0.2222004383802414, 0.5299170613288879], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.25, 0.07, 0.68], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1941285236e2e0041140a0e906599ce2fca651f884eb734d95a8566f29c722a1:action", "state_id": "6ae86bbde3f4b107de28e17c22e62aa173c0f5971f6ccaf99014eac2c093f7fd", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.15, 0.08, 0.77], "teacher_probs": [0.15, 0.08, 0.77], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.45556640625, -1.015625, -0.01953125], "student_probs": [0.3207452893257141, 0.1832018792629242, 0.4960528612136841], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.15, 0.08, 0.77], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7e7de9779f13c7036044673a857ac97fd77a9d515d87d1dc5bd5ddd5413c92a1:action", "state_id": "32b0ff3c79ba6c9efe44e3e8efe280640258cddb6ab37ee1c6c789a55c77dd19", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.6, 0.15, 0.25], "teacher_probs": [0.6, 0.15, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.228515625, -0.05859375, -0.33984375], "student_probs": [0.3246903717517853, 0.38482698798179626, 0.29048264026641846], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.6, 0.15, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ed611366d39a341540de872e36b8926d0093341c8e326f207a059c76dd278051:action", "state_id": "171a184e1d2f3daa5340ce1de33a0979763338ca47b896173e4b483bae8a7e8b", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.58, 0.42], "teacher_probs": [0.58, 0.42], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.2578125, -0.287109375], "student_probs": [0.507323682308197, 0.4926762878894806], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.58, 0.42], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "be47c63b8f3f79528dbb43ea4ed71ddbb4fb163c4d607d5d0999eedd62423997:action", "state_id": "0f5b8eee86d7a4e8a7ce5f3a1421c748c4858f63b23bf52b42970ef8b7929c6d", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.46, 0.31, 0.23], "teacher_probs": [0.46, 0.31, 0.23], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.86328125, -0.349609375, -0.5693359375], "student_probs": [0.24918220937252045, 0.4164874255657196, 0.33433037996292114], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.46, 0.31, 0.23], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fa37692cbd84df2e66ef92dc8790ff5175c80906e292451bcf8c51b06d5b2011:action", "state_id": "37975e8a69916a7104bb1bf964560a54827210daf74e809aa3f37c2673e87412", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.74, 0.11, 0.15], "teacher_probs": [0.74, 0.11, 0.15], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.08984375, -0.5439453125, -0.26171875], "student_probs": [0.4036974310874939, 0.2563552260398865, 0.33994731307029724], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.74, 0.11, 0.15], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f17c1af8b976879a4d6e2b2d680a3d058ef7631aaab08001ffb38adb28191a20:action", "state_id": "9c27bfa6a3b4c1183a411612fe8f0d82f918a7e984e32936b9ffc2a5454f2818", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.65, 0.16, 0.19], "teacher_probs": [0.65, 0.16, 0.19], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.48828125, -0.6536865234375, -0.2265625], "student_probs": [0.3177921175956726, 0.2693447470664978, 0.412863165140152], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.65, 0.16, 0.19], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0a140dca29c980a84857cda5aba3c9a5b8c5d33c571c1c15efbb40598dd89b04:action", "state_id": "bb60de36c2241f6d7994152bae855008f2567b46f1dec6ab7bb419313e5b9f97", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.17, 0.83], "teacher_probs": [0.17, 0.83], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4853515625, -0.5009765625], "student_probs": [0.5039061903953552, 0.49609383940696716], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.17, 0.83], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9adcadfd869a6ce30b2961dee5a69a2f1453fef75dc7616c85597ab6da8ca795:action", "state_id": "f690e8be2bc00f1dd9a3422bfea0922e4c3f486c28dbeb863e6cc4f05bb8159c", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.68, 0.32], "teacher_probs": [0.68, 0.32], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.328125, -0.361328125], "student_probs": [0.5083000063896179, 0.4916999638080597], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.68, 0.32], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e0c37ab6c53915edaff382cbc8833909bc561525fa230526f3c898c280d90999:action", "state_id": "e320e7ff4864c0734ca5aabfd45592bbec7c5ef619852fe6d2a885d888de3fa3", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.71, 0.29], "teacher_probs": [0.71, 0.29], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4775390625, -0.6473388671875], "student_probs": [0.5423482656478882, 0.4576517641544342], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.71, 0.29], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "121d554597332f647b1ae4190391404cd5ce7bb560f5190a8a1114455c2775d1:action", "state_id": "50ab3d9d37fd48e778623dbba7dd40b3a5b9670ebd2e528fe082d644bd2f9830", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.28, 0.14, 0.58], "teacher_probs": [0.28, 0.14, 0.58], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.03125, -0.888671875, 0.0390625], "student_probs": [0.40046486258506775, 0.16989900171756744, 0.4296360909938812], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.28, 0.14, 0.58], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "37d331cb3269f9e330d27ee33a6e82774fd04a26be260658a947b43e22dd6af5:action", "state_id": "403d72e0f8e9e74ca6aad9683a87f1c5f4aaa1fc9967598254a832ed2cc2c7a3", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.15, 0.3, 0.55], "teacher_probs": [0.15, 0.3, 0.55], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4990234375, -0.236328125, -0.3046875], "student_probs": [0.2845003306865692, 0.3699728846549988, 0.3455268442630768], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.15, 0.3, 0.55], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fedf4286ba01156e66ebc251a00231a676e611019b4087bc97dfc0eaf454bc00:action", "state_id": "52a1b34f7dc0979b6bf519ccd7a86153998e58c86b3eb25dc33dd71e8fc075d9", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.24, 0.15, 0.61], "teacher_probs": [0.24, 0.15, 0.61], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.263671875, -0.30078125, 0.01953125], "student_probs": [0.3038640320301056, 0.29279449582099915, 0.40334147214889526], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.24, 0.15, 0.61], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5b2914c68c92a6c39a75df8ede0ae02565a80dd055edc740638264c4479e3ab2:action", "state_id": "f7597c546b159b2aeb41b195acd23b758ad6881ae5480f0e59eb61e0433c6679", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.33, 0.22, 0.45], "teacher_probs": [0.33, 0.22, 0.45], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.34765625, -0.31640625, 0.03125], "student_probs": [0.2863336205482483, 0.29542282223701477, 0.41824355721473694], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.33, 0.22, 0.45], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "32e4d89a57dc9e50cea2283a089dc1e7662e43e8dcefd50850ab3414c00ae99c:action", "state_id": "1be3ec722d8a7a190a62b115220e81cf01398fd1f75ac7425c70829b5d8f9cd9", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.27, 0.23, 0.5], "teacher_probs": [0.27, 0.23, 0.5], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.322265625, -0.2890625, 0.21875], "student_probs": [0.26655927300453186, 0.2755584120750427, 0.4578823149204254], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.27, 0.23, 0.5], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "147730e88c95c3ba1c8571f0813e9a0bd4e2c79c86e42f58f8738b31fecd9b15:action", "state_id": "3d33fc831372987396fcb9a7fdd85ab828521082bae7a9e2591be6a058e08956", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.29, 0.28, 0.43], "teacher_probs": [0.29, 0.28, 0.43], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.544921875, -0.53271484375, -0.04296875], "student_probs": [0.27290889620780945, 0.27626070380210876, 0.4508303701877594], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.29, 0.28, 0.43], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b93371d0093972a67dd046df70e777dda985b94bf4bd68e95096cb4f657b3063:action", "state_id": "40eb167fe6458a9f9e09824fa50ef1efdd4f44068bc3d6f830558d160877e954", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.33, 0.5, 0.17], "teacher_probs": [0.33, 0.5, 0.17], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.423828125, -0.3525390625, -1.1953125], "student_probs": [0.3942878842353821, 0.4234224557876587, 0.18228965997695923], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.33, 0.5, 0.17], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "b777090a3a2574ddddc1e06298f9df71c94dc278cf96489e951e5768159df428:action", "state_id": "a9e97e5f4087a8198f22ff8a2789ec912d9e4ae5830bdc11556d0b434d3badc0", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.59, 0.15, 0.26], "teacher_probs": [0.59, 0.15, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.31640625, -0.220703125, -0.23046875], "student_probs": [0.31346285343170166, 0.34494465589523315, 0.3415924310684204], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.59, 0.15, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f7cc8b96ab2689d6a747b12baf9e87e43a05f7429a6ebf91a6017a8b83d25265:action", "state_id": "fb06a1754f9136213229fe5881baa01239ea799bea2bc1c9c8cd510416110482", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.68, 0.11, 0.21], "teacher_probs": [0.68, 0.11, 0.21], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.134765625, -0.21484375, -0.205078125], "student_probs": [0.3502447009086609, 0.3232913315296173, 0.3264639675617218], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.68, 0.11, 0.21], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4e96d4589eb4a1de4cd740e1e1cc32f1f11c05bff3c3326a143a712743ac6d68:action", "state_id": "b0168d6267841e07d78a10498f584c58dcb663a907f927abffa5976cd975409a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "west"], "gold_index": null, "teacher_raw_probs": [0.48, 0.52], "teacher_probs": [0.48, 0.52], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.5361328125, -0.1953125], "student_probs": [0.4156102240085602, 0.5843898057937622], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.48, 0.52], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e56d1e6a9339ffa2cb25f432d8f5db3f2c2c9e75ee3dbd49d7684500cfb69af7:action", "state_id": "9d69abcb8b3d5b254b4d35e5fccc48eba36ebd26894a37d58495b43e5063b484", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south"], "gold_index": null, "teacher_raw_probs": [0.47, 0.53], "teacher_probs": [0.47, 0.53], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.314453125, -0.349609375], "student_probs": [0.5087881684303284, 0.491211861371994], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.47, 0.53], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5c25fed2d82a9dfbedc31be74f3e1fe14fe15e3ca90a207e1bbd30edfd813474:action", "state_id": "1d857e496ee2b4b9f6b0cf70dfd4eff8a42f9535c1dc76e3f1741f0e370de1ac", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.36, 0.27, 0.37], "teacher_probs": [0.36, 0.27, 0.37], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.07421875, -0.26171875, 0.0], "student_probs": [0.34410718083381653, 0.285274863243103, 0.3706180155277252], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.36, 0.27, 0.37], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "da0c56c6eb65a1f6988f33f706dfe1b89f8065c537a8c1f9d7f1abbcbd2bb336:action", "state_id": "af2f950aeaf98d88da09547cf6df15a34feb48a3eca01198eac9968cdc7568b2", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.29, 0.71], "teacher_probs": [0.29, 0.71], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.357421875, -0.134765625], "student_probs": [0.44456472992897034, 0.5554351806640625], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.29, 0.71], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "acdd8bf5c744f90466debd59dcb2b9d137c2008643af5f1676cf6d972a69077c:action", "state_id": "2e66a54ef170f096c43f72c32a22fc09b9fc8ca556ad7ebe65bc1b5fc24af3ea", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.35, 0.32, 0.33], "teacher_probs": [0.35, 0.32, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.251953125, -0.05078125, -0.4189453125], "student_probs": [0.3258346617221832, 0.39844200015068054, 0.27572330832481384], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.35, 0.32, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "62c3394f8b056f146115de8d08ef1ca3e8ecf5df569b440706d0276b26037d10:action", "state_id": "fcfa64b8aab27146ed01f7b4e5b08cc4e3e2610725d618da76877065faeef05a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.37, 0.63], "teacher_probs": [0.37, 0.63], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.337890625, -0.56494140625], "student_probs": [0.5565201044082642, 0.44347989559173584], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.37, 0.63], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a6d0f0c6b1b7506318188f19fc8117b2b82fbb851505a269eab9af6e26923ee7:action", "state_id": "c65a96d44ac13938e160a50e645bf0355239b658acf5b8041eda8ddb4d878c50", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.21, 0.59, 0.2], "teacher_probs": [0.21, 0.59, 0.2], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.6995849609375, -0.0390625, -0.322265625], "student_probs": [0.22757409512996674, 0.4405387341976166, 0.33188721537590027], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.21, 0.59, 0.2], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6105348a7881a17135ee59ab1f5226eda1a1492568351719508e12e0493b0279:action", "state_id": "635367a6e4adbba27f73e6a7186ac9a6224c65c6f4efc9596997c6faad75ba57", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.24, 0.76], "teacher_probs": [0.24, 0.76], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.41015625, -0.236328125], "student_probs": [0.45665207505226135, 0.543347954750061], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.24, 0.76], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "be278c49acc7be4568eda1a0023e4042bd2a32f307b4a76407033314448e87a3:action", "state_id": "893148370307b3776c27d6bcf6b47ae97122fd6dc908061a54b702409fd3ac44", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.23, 0.59, 0.18], "teacher_probs": [0.23, 0.59, 0.18], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.263671875, 0.06640625, -0.224609375], "student_probs": [0.29146766662597656, 0.4054539203643799, 0.30307841300964355], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.23, 0.59, 0.18], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "87fafd5d706bbb599da7511abd061d0e36dc256531e48bc39bbe7dc3a2b819c9:action", "state_id": "b150982c6597ca39e3543304515d4626a1c7ba1e3dbf52d953508352333a0658", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.69, 0.13, 0.18], "teacher_probs": [0.69, 0.13, 0.18], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.0703125, -0.08984375, 0.18359375], "student_probs": [0.3364785611629486, 0.2866833209991455, 0.3768381178379059], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.69, 0.13, 0.18], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a0d94041968effeeef6172e418c66c7c849b53272a42540b48c841a711efe244:action", "state_id": "1c948a06b2913c305bc02d1d7a3f2257f061197560395894e2353d399afe7eef", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.59, 0.2, 0.21], "teacher_probs": [0.59, 0.2, 0.21], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.09375, -0.1328125, 0.1875], "student_probs": [0.3453569710254669, 0.27534258365631104, 0.37930044531822205], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.59, 0.2, 0.21], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "907d4475b1d1e8a9a295502928d83698a63bd56e368323cc14c860d85c722c74:action", "state_id": "29ecc972c1572fec86d5bce647c6a8a06e90cdece61a90fe141e0f484117482e", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.3, 0.42, 0.28], "teacher_probs": [0.3, 0.42, 0.28], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.21875, -0.05078125, -0.5400390625], "student_probs": [0.3438655734062195, 0.40675845742225647, 0.24937596917152405], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.3, 0.42, 0.28], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9c1871db0ed88d50592f9272d4212431c82839568061273efb8905f68b92055b:action", "state_id": "c64dcd23601f93f48dd39f83ea390c9c38ca5c741307955b50658764dcc42e7a", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "west"], "gold_index": null, "teacher_raw_probs": [0.39, 0.61], "teacher_probs": [0.39, 0.61], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.13671875, 0.0390625], "student_probs": [0.4561675190925598, 0.543832540512085], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.39, 0.61], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "cbda5e7fa603305c474ce4351caba47cfa6ff2ea10802d3d026dce48579ac2e4:action", "state_id": "2657e991d3e76cdcdc050aa9ed22fa8bc40273e4f7df980b8236d3c0ecfaaa15", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.35, 0.27, 0.38], "teacher_probs": [0.35, 0.27, 0.38], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.169921875, -0.587890625, 0.171875], "student_probs": [0.3261730670928955, 0.21474674344062805, 0.4590802490711212], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.35, 0.27, 0.38], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "74880447dbcb3341d69bfec53d15a03c503c86a1bf5232abaf7838cba71cedaa:action", "state_id": "057e31502d6b19237be68a260ac909a6d26e1fe41e84479aba091b27976f4763", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north"], "gold_index": null, "teacher_raw_probs": [0.5, 0.5], "teacher_probs": [0.5, 0.5], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.396484375, 0.01171875], "student_probs": [0.3993430435657501, 0.6006569266319275], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.5, 0.5], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "82c12e3e324a571a74cca592372f0a540db138a44ca3a1c7a80ffa31df75e21b:action", "state_id": "9c0874e90469b8259540253fee2085d42b3730084c5c28c663a0de1144f308f1", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.26, 0.34, 0.4], "teacher_probs": [0.26, 0.34, 0.4], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.5146484375, -0.310546875, -0.380859375], "student_probs": [0.29677340388298035, 0.363969624042511, 0.33925700187683105], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.26, 0.34, 0.4], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "18f851302e1249d844f40a3b6f3b1a8c9b694a13e76fc8a1637c8dbe781d5a7f:action", "state_id": "cb82502d69ea0b1b6fb6d263891cec72fd10e3d87fd7a5bbda5712bc068bc580", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.2, 0.5, 0.3], "teacher_probs": [0.2, 0.5, 0.3], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4892578125, 0.0, -0.400390625], "student_probs": [0.26852551102638245, 0.4379933774471283, 0.29348108172416687], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.2, 0.5, 0.3], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a285c9cb1053a45be48a2816b467f2fb3fc96967cbb34f7a01fad8fbc771f294:action", "state_id": "d894b0600ef470323a124b00a9c09f94e49c5b787eea473ad12b5c2f93e436ac", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.2, 0.53, 0.27], "teacher_probs": [0.2, 0.53, 0.27], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4697265625, -0.01171875, -0.4697265625], "student_probs": [0.27925771474838257, 0.4414844810962677, 0.27925771474838257], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.2, 0.53, 0.27], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "751ed9a0789ba8ff65777ec264ab4c67f3e75950f0971a0c0ca3b9140705be72:action", "state_id": "1d6da4f2ceb943934d1c7149e0c2a13df4751e14a43c858604a6b16198ce5c46", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.18, 0.41, 0.41], "teacher_probs": [0.18, 0.41, 0.41], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.12109375, 0.4921875, -0.08984375], "student_probs": [0.2578499913215637, 0.4761149287223816, 0.2660350501537323], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.18, 0.41, 0.41], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8a6c5b8a158e68fa534be1f9e0344727f6d86d7da74b81afc4b87b63af29fb9f:action", "state_id": "24428bbbfa2640c0ef4c3f98185f5847b0c939c702b3330c7ce06a35a1c3c386", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.2, 0.36, 0.44], "teacher_probs": [0.2, 0.36, 0.44], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.3515625, 0.572265625, 0.42578125], "student_probs": [0.3008427619934082, 0.3751368224620819, 0.3240203857421875], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.2, 0.36, 0.44], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "85b98deb717b2849c44142f70d7a4ff1d4255f36f135fc15d39edaf5e5b1aadd:action", "state_id": "280ac21d8c8c5c90aa5df99ffc31eae98fc16d98d88fe8dd67b945f105f85668", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south"], "gold_index": null, "teacher_raw_probs": [0.48, 0.52], "teacher_probs": [0.48, 0.52], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.388671875, 0.03515625], "student_probs": [0.3956010639667511, 0.6043989062309265], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.48, 0.52], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "9d0ecb29ab31e89fa21b5a3683ecac92789ba21d108b57db65bccaaa7efd59d8:action", "state_id": "7a7af1cf5361291844b1e99408f3bb5dacfdf0f0f0a395af5bc4eb2aa503ab11", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.29, 0.43, 0.28], "teacher_probs": [0.29, 0.43, 0.28], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.685546875, 0.708984375, 0.45703125], "student_probs": [0.3546818494796753, 0.36309289932250977, 0.28222528100013733], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.29, 0.43, 0.28], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0bf458ebbda937c60db9c584503e215b5261edb0bdde2738878d1aba4d343bda:action", "state_id": "dc8728ad30ec0bced275e3adddb2232b1ff08da596b3744b307da4dd18197d53", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.68, 0.23, 0.08], "teacher_probs": null, "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": null, "student_logits": [0.125, 0.03125, -0.0234375], "student_probs": [0.36067694425582886, 0.3284001052379608, 0.3109229505062103], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": "Rounded target has no identity probability-simplex representative", "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": null, "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7e58fd03b85d98925bd5474139ae9e79810ccff8dda174e34cbdc57ce304f434:action", "state_id": "687aa1f30f805e0b1a166393961ab99e064416b363618738d621b7744c950688", "family_id": "unified_maze", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.34, 0.54, 0.12], "teacher_probs": [0.34, 0.54, 0.12], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.046875, 0.41796875, 0.140625], "student_probs": [0.26329678297042847, 0.41910672187805176, 0.31759655475616455], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "maze", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.34, 0.54, 0.12], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "a1c26323fb3429dfc5a51b91c79df95cdfb80bbc287277c9835b56a99d9cf03f:action", "state_id": "2d20983affddeccaea9d24acc2bd32e506f9747a1caafe2281b0b0f56e29be27", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.75, 0.11, 0.14], "teacher_probs": [0.75, 0.11, 0.14], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.203125, -0.3984375, -0.255859375], "student_probs": [0.6973240375518799, 0.14056748151779175, 0.16210849583148956], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.75, 0.11, 0.14], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6a1cd23727139daa0187f897277ab0e0dc54005fabf1cf7a773a70bf8bd5be9e:action", "state_id": "28865b9658718e77377887bd4d88870fdf3a3c66f6ea11fcf3907f91d42bc07d", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.75, 0.19], "teacher_probs": [0.06, 0.75, 0.19], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.680419921875, 1.951171875, 0.26171875], "student_probs": [0.0572693906724453, 0.7958081364631653, 0.14692246913909912], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.06, 0.75, 0.19], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "187636a56982cc0a3c7c08b0a3472f27f0135c3b3da20c05208ab0d26622bfcc:action", "state_id": "6d41ecedc038569196aed370d3fa7f20a352356dcae92e466a8b479c58ab1df7", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.55, 0.41], "teacher_probs": [0.04, 0.55, 0.41], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.3046875, 2.05078125, 0.890625], "student_probs": [0.06735068559646606, 0.7100829482078552, 0.2225664108991623], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.04, 0.55, 0.41], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1dc60275754c86c7dc32afc3081531ecbbd11ea45b56145ad064e8138ba7e38f:action", "state_id": "d7ad8484efba18eb569e34adee7382d92e965bf0f5aa6c86094625cfd020f5d3", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.0, 0.04, 0.96], "teacher_probs": [0.0, 0.04, 0.96], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.185546875, 0.3359375, 3.748046875], "student_probs": [0.01859607733786106, 0.031325578689575195, 0.9500783085823059], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.0, 0.04, 0.96], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "93c55ae01724fb47ae209b0320d55a10d995c9f0916a481c4b3ecfa79a84fbf1:action", "state_id": "35da4b37d898b07669b41fe07921dbb072610232cc7c55c05192cae31a23a675", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.36, 0.59], "teacher_probs": [0.05, 0.36, 0.59], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.41015625, 0.177734375, -0.07421875], "student_probs": [0.10312493145465851, 0.5046331882476807, 0.39224186539649963], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.05, 0.36, 0.59], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "0c7c84da13768c35e0ddb993b05e1e3a9b0de349653a354a368d201dee9e43ec:action", "state_id": "8fdc8644fa01b4e19af8360b541ff1679cee53beef20d56051177d910d6a9b54", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.92, 0.01, 0.07], "teacher_probs": [0.92, 0.01, 0.07], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.08203125, -1.2265625, 0.55078125], "student_probs": [0.7981911301612854, 0.029187902808189392, 0.17262093722820282], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.92, 0.01, 0.07], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ff963fc724fd47c644c64970454f34e596e3ac65eb51dec714fe9854bf9967a1:action", "state_id": "976acb34ea5463f76c343bc659adbe2b6978bd89c93a61acfd674724dbfe09cd", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.61, 0.33], "teacher_probs": [0.06, 0.61, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.7890625, 0.146484375, -0.53076171875], "student_probs": [0.08735709637403488, 0.6051952242851257, 0.3074477016925812], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.06, 0.61, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f5004680ab96105932aaa744378f5b5a490a59e2c5e3bda7ea1c2ac3b5728295:action", "state_id": "0fcae231aa2a1fef0218eb9a89182b95d0917fc67c26483b9956f83b7b778442", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.1, 0.58, 0.32], "teacher_probs": [0.1, 0.58, 0.32], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.234375, 0.669921875, -0.2265625], "student_probs": [0.09565453976392746, 0.6422901153564453, 0.26205533742904663], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.1, 0.58, 0.32], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "711d6f8f01a168c7d8e16cf2f67256795a24bf085d7100d206f4da8cc2850b3a:action", "state_id": "ad53bfd682c8d39a0c7d8c297aa7a8d52a1cb72972604628995536d65ff472af", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.08, 0.5, 0.42], "teacher_probs": [0.08, 0.5, 0.42], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.921875, 0.6171875, 0.662109375], "student_probs": [0.09492567926645279, 0.44237443804740906, 0.46269986033439636], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.08, 0.5, 0.42], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "ebfe9bb3911d957ce66420ff974ff5be5e3e07a2c485ebe6509487418746afca:action", "state_id": "4efd532ca67c776526c0bb3a1ed896c480503848db36e60bc3ca18f676a37182", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.37, 0.58], "teacher_probs": [0.05, 0.37, 0.58], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.88671875, 0.447265625, 1.29296875], "student_probs": [0.07331549376249313, 0.27831578254699707, 0.6483687162399292], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.05, 0.37, 0.58], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "fe47f917e44915959650fe6a5443787a96a141d80c869bd4f9892e395f317c76:action", "state_id": "b7e8ced96070a3e9e079c8ffab966e727b04d5b1d40423038536c1a931412534", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.09, 0.04, 0.87], "teacher_probs": [0.09, 0.04, 0.87], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.5390625, -0.2734375, 2.767578125], "student_probs": [0.09319821745157242, 0.04135645925998688, 0.8654453754425049], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.09, 0.04, 0.87], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "65953e699660bbc221730bedff8286d886cb778904d7708b10847846a94d129a:action", "state_id": "6fd0cf2b21fe60217a71b4cc7940179eab8f752b0d06063cc2052a7236ae358a", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.08, 0.03, 0.89], "teacher_probs": [0.08, 0.03, 0.89], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.69140625, -0.4599609375, 2.60546875], "student_probs": [0.12350583076477051, 0.039053063839673996, 0.837441086769104], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.08, 0.03, 0.89], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "44508dae09a4c81afb692d2c69508c0afa0ec965d2ea21786643b58f7eb5698e:action", "state_id": "3c8cb6d03a39552fb5de2c8e3934aba9ef75ab62f2f13f80c4655a4c8238966c", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.06, 0.02, 0.92], "teacher_probs": [0.06, 0.02, 0.92], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.53125, -0.03515625, 3.005859375], "student_probs": [0.07437914609909058, 0.042214736342430115, 0.8834061026573181], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.06, 0.02, 0.92], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7cc0829deaa737fad12e9ec7456f1d7d5a5e7a784aafda0742e1cf7e2f93bca6:action", "state_id": "a7d2e9be81991cf0711906cbd18ce55711328aadcd3dd05d79d499a508d2f73a", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.01, 0.96], "teacher_probs": [0.03, 0.01, 0.96], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.3125, 0.09765625, 4.297119140625], "student_probs": [0.017994843423366547, 0.014515855349600315, 0.9674893021583557], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.03, 0.01, 0.96], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "c533015f6d5db1230b7005df44d20cb7d7df6546852f8efd1aeb30a35d6f6218:action", "state_id": "b4c979289c9f62aed03f1e78e65c1b008c0264b5c74e70865fac5ed5a4994e6e", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.58, 0.38, 0.04], "teacher_probs": [0.58, 0.38, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.3671875, -1.1953125, -1.66796875], "student_probs": [0.34155696630477905, 0.4056089520454407, 0.2528340220451355], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.58, 0.38, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "538dd39369ac10a5b8815320da3535e675a440378740fcd7329cc6ba53abb197:action", "state_id": "9f2e7508a60cba421e7ddf2d53a02bbaf8820d702fa6c32195009bccbf385a40", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.5, 0.49, 0.01], "teacher_probs": [0.5, 0.49, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.884765625, -0.87109375, -2.67578125], "student_probs": [0.4585985243320465, 0.46491149067878723, 0.07648996263742447], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.5, 0.49, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "7055874a6fe3c3cdeeaebd5fbc9c50f3a6e8bfc4cd7bd8cf15dfaafa81f49d33:action", "state_id": "a22abb7eaf42d252f6cd47f197855d08c9a13ccea182e87a7e922fe4a29df084", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.67, 0.17, 0.16], "teacher_probs": [0.67, 0.17, 0.16], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.2734375, -0.53125, -0.30859375], "student_probs": [0.7298827767372131, 0.12008459866046906, 0.15003260970115662], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.67, 0.17, 0.16], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "60644f41e885c722f07f02462b62f6f1ff38172d97b4ddfa9a87eae7e693ac60:action", "state_id": "8e23b13c7826edad5ca436b1f41937ef04c5f93d462d938e135f22b8a01fb158", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.25, 0.63, 0.12], "teacher_probs": [0.25, 0.63, 0.12], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4052734375, 1.046875, -1.19140625], "student_probs": [0.17458446323871613, 0.745874285697937, 0.07954125106334686], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.25, 0.63, 0.12], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "2a186a3bb8fd988e023c9088ff5353857b47d14173ac51be42855cc1c0027860:action", "state_id": "3df6f41d60d018e4915c37ee83cd6fc47be954d63c469c778323142c0332895f", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.37, 0.56, 0.07], "teacher_probs": [0.37, 0.56, 0.07], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.189453125, 1.412109375, -1.2109375], "student_probs": [0.21539202332496643, 0.7315137386322021, 0.053094275295734406], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.37, 0.56, 0.07], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "db12abfbd2d302d495427a5b7cc4761f4d443a0c08bca46d17fea9e50a5107ed:action", "state_id": "45b960955dab1894bb651921344ffdfaabeac58efa284c56e849368918c46729", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.91, 0.06, 0.03], "teacher_probs": [0.91, 0.06, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.1328125, -0.01953125, -0.78515625], "student_probs": [0.8545148372650146, 0.09930442273616791, 0.0461808480322361], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.91, 0.06, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "06109d09565c43961bdf3b0fc9a17f33f124059beb8d76842b0e96a19a1c33c0:action", "state_id": "d0f561d2d68b4fb77626e2228e7cc3d6a34ee3e8a5645be4ea1801038046a679", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.9, 0.08, 0.02], "teacher_probs": [0.9, 0.08, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.345703125, 0.05078125, -0.7880859375], "student_probs": [0.8738800287246704, 0.08806025236845016, 0.038059625774621964], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.9, 0.08, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "06bcae81211ea3e2d4d0dcf19754ba5c618f4047757d000f6742f1495ba53c34:action", "state_id": "4ea6b13e3252f74b25611befb42c9d4be60e66f13891cefb95ce9fd98a4ecb5e", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.93, 0.06, 0.01], "teacher_probs": [0.93, 0.06, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.59765625, 0.123046875, -1.12109375], "student_probs": [0.9021523594856262, 0.07595750689506531, 0.021890077739953995], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.93, 0.06, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "20f5422c079ae0744411a2a162d7099a6834ce3fdf4a4b5e0a07edce7387ad33:action", "state_id": "659e860e5e4a8b0bfd09c8353cf52cc5bbb344540673e3998f35f96867675049", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.95, 0.04, 0.01], "teacher_probs": [0.95, 0.04, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [2.248046875, -0.17578125, -1.359375], "student_probs": [0.8962954878807068, 0.07939552515745163, 0.02430903911590576], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.95, 0.04, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4706df15f4cc6027e47ea88ecea645ccb4e1f76fd0a43edd892105a853cf7230:action", "state_id": "46df381ee65cc98beee090260d5a9ab99276e8ea100265c6fdaf5cfcd472d4b1", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.99, 0.01, 0.0], "teacher_probs": [0.99, 0.01, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [3.720703125, -0.072265625, -1.39453125], "student_probs": [0.9722583293914795, 0.021903639659285545, 0.005837993696331978], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.99, 0.01, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "00f95d897c8c054e718ce3db0a30da9f2decd8d0e3f51ac7c5f7ebc4a7d5560e:action", "state_id": "c943323d0f2a712229855f4872af3506fb1c3ddcbd0bfa9dc473eea383866f12", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.79, 0.19, 0.02], "teacher_probs": [0.79, 0.19, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.716796875, 0.05859375, -1.7578125], "student_probs": [0.6242287755012512, 0.32321372628211975, 0.05255748704075813], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.79, 0.19, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "1a1684ca1b938022e588f2db29c79db0dade6bdf127d9c947c425ed1fbe76ab6:action", "state_id": "03aa3064e6ba696ba7d25ca5d51053ae32210d3ec9d091f44f7c64e8310c69c8", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.73, 0.25, 0.02], "teacher_probs": [0.73, 0.25, 0.02], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.34765625, 0.494140625, -1.44921875], "student_probs": [0.6725332140922546, 0.28644195199012756, 0.041024789214134216], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.73, 0.25, 0.02], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "783902ec47316bba3618f7543dc98132b12839e4ab8b48a5ba67c57ad9ff3937:action", "state_id": "0c3e0c07e774cbbec0196578dc4e153f0a81ae1cfdce8ff187f9429bdf855cec", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.61, 0.36, 0.03], "teacher_probs": [0.61, 0.36, 0.03], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [1.298828125, 0.703125, -1.20703125], "student_probs": [0.6124522686004639, 0.3375683128833771, 0.04997943341732025], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.61, 0.36, 0.03], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6fafa97190c75ce92907c37ee715d582fb0f071c942cee10f98455e9f177d253:action", "state_id": "354e672c361bb07f03d570187358da0c408d7d6ea09a45980bebe2936537e058", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.43, 0.53, 0.04], "teacher_probs": [0.43, 0.53, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [0.4140625, 1.400390625, -0.921875], "student_probs": [0.2535315155982971, 0.6798120141029358, 0.06665638089179993], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.43, 0.53, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "12c11ac0a1851ff94e13ced6f2989f96ac9594df317b3585169c46cbe53fe7cc:action", "state_id": "cd5788a923d73eb724847598c53dd80e8418ce43529a56d4b1474073c5eb8c8e", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.9299999999999999, 0.04], "teacher_probs": [0.03, 0.9299999999999999, 0.04], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.3984375, 1.927734375, -1.0546875], "student_probs": [0.03306679427623749, 0.920301616191864, 0.04663165286183357], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.03, 0.9299999999999999, 0.04], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "f2cfc8337e545a1fa8a98dd821269cb8e083dcea0206f65d4a508298068fc300:action", "state_id": "4d8e0f9dbc0574ec40986c55f7548a58f02b49d5d89905cf6d46fb23a3f9b07d", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.98, 0.01], "teacher_probs": [0.01, 0.98, 0.01], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.23046875, 2.29296875, -0.974609375], "student_probs": [0.027630161494016647, 0.9366835355758667, 0.03568631783127785], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.01, 0.98, 0.01], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d8a4ff14bb614210e9f60b843acc064cf8587536e3945c3c796f5102798dbc4d:action", "state_id": "d5f977b2caac558a63c4212386b498407179c0f7ccbb40e5c8214ed528fdae64", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.0, 1.0, 0.0], "teacher_probs": [0.0, 1.0, 0.0], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.859375, 3.94140625, -0.48486328125], "student_probs": [0.00806063786149025, 0.980216920375824, 0.01172243244946003], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.0, 1.0, 0.0], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "03a417af26b4ff2fb5458f78330eeb0806c054b26cbf9fa913e8b8f695cb7590:action", "state_id": "20dfe6f5b23bde3ed03df8ca874ee110f01ec0d13fdcfe2d99c1ca7e7d2c1264", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.43, 0.55], "teacher_probs": [0.02, 0.43, 0.55], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.896484375, 0.716796875, 2.34765625], "student_probs": [0.031586676836013794, 0.15854153037071228, 0.8098717927932739], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.02, 0.43, 0.55], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "8fde7751a88974e4684987942fde86aebeb3519f7c180c62bb60fa66d967e063:action", "state_id": "cd4b14c69c2c9641f0f2b235edd7621c6945156a2fc4fd36ac089730c8faebd8", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.93, 0.01, 0.05], "teacher_probs": null, "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": null, "student_logits": [4.1055908203125, 0.140625, 1.302734375], "student_probs": [0.9262644052505493, 0.017570016905665398, 0.056165535002946854], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": "Rounded target has no identity probability-simplex representative", "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": null, "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "32aada99fdcc866259b12e721e9fcc44893e95d587883a68c54982b2b50ae119:action", "state_id": "968515504effc10e6b30781030d0c25a3a440d58ff952877d1f36bd49c8b5c2b", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.35, 0.46, 0.19], "teacher_probs": [0.35, 0.46, 0.19], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.14453125, 0.0859375, -1.0546875], "student_probs": [0.37570658326148987, 0.47308602929115295, 0.15120738744735718], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.35, 0.46, 0.19], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "682ae21b32376787d358c24a6eaf1169bb9edea79e4dc166e2a7fe9b8268e984:action", "state_id": "a9f25dfd5efdbd1787a4d2bafd898a2fecd14d5fee9c9695da627336a3da2e39", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "west"], "gold_index": null, "teacher_raw_probs": [0.65, 0.09, 0.26], "teacher_probs": [0.65, 0.09, 0.26], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.615234375, -2.03125, -0.7421875], "student_probs": [0.4709308445453644, 0.11428503692150116, 0.41478410363197327], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.65, 0.09, 0.26], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "107139c821975e2d002026f701aea24c6755eb0d62bf7442b06c08e1d5a6aeda:action", "state_id": "bf7f2ed904da0c63ceb963ad0be44cd64a771dd5c9d7775c0c5049f33b8f94a8", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "north", "south"], "gold_index": null, "teacher_raw_probs": [0.56, 0.03, 0.41], "teacher_probs": [0.56, 0.03, 0.41], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.2705078125, -2.8203125, -0.3212890625], "student_probs": [0.492954820394516, 0.038498252630233765, 0.46854692697525024], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.56, 0.03, 0.41], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "11cbc03aa902843368cf17b0f0f841773e6924b920dd967f20859db0c4855b0c:action", "state_id": "08cb978e69a55965bd0c070236bd80523bf05dc3b9501e61619fef3bdd6e7418", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.16, 0.6, 0.24], "teacher_probs": [0.16, 0.6, 0.24], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-1.2890625, 0.525390625, -0.595703125], "student_probs": [0.1094314232468605, 0.6716592907905579, 0.21890929341316223], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.16, 0.6, 0.24], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "72f00737ffc434f9ed0a9b9356bb683795dedaf7bbc296a6043ac638ef38210d:action", "state_id": "912f0d45a1112013c0a880bcf5fc90a9dd6721e6bb785918e6de420364946731", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.12, 0.64, 0.24], "teacher_probs": [0.12, 0.64, 0.24], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.689208984375, 0.9921875, 0.140625], "student_probs": [0.11539360880851746, 0.6200160980224609, 0.26459038257598877], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.12, 0.64, 0.24], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "26ccb1a1b2c524a8599263f7ce3e3c60f47a62d613619bd1529718f367a702f1:action", "state_id": "b45beef2a45473ce87c50f656790ccde917a4625912b67eaafbb003bf661266f", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.05, 0.61, 0.34], "teacher_probs": [0.05, 0.61, 0.34], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.7958984375, 0.984375, 0.21875], "student_probs": [0.10320053994655609, 0.6121317744255066, 0.2846677005290985], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.05, 0.61, 0.34], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "dde3cdc92db2f8067244ea9a3e10e94e7dc7e8d33f9494446486eae2e8efe86f:action", "state_id": "29be622b7c96a638fad90d096348edc56bc64cb9f814edd99711fe58336a8472", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.63, 0.33], "teacher_probs": [0.04, 0.63, 0.33], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.678955078125, 1.07421875, 0.501953125], "student_probs": [0.09969863295555115, 0.575549840927124, 0.32475149631500244], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.04, 0.63, 0.33], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "93b3bc111b44c4a99cffb7c68b9e0a50125b2e9a9c4ad62f5812b89128dc7200:action", "state_id": "dca1534538e910f3b412c914e678c8e56912542510e532b55e29e2730a93c1b0", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.03, 0.61, 0.36], "teacher_probs": [0.03, 0.61, 0.36], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.64349365234375, 1.111328125, 0.556640625], "student_probs": [0.09898070245981216, 0.5723477005958557, 0.3286716341972351], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.03, 0.61, 0.36], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "447a144a40257bdc271dd43c3b85d38d726a2eddb99c7be30634858530925089:action", "state_id": "94afcbc6fda82da67fa2b648b8c9504b33c3588903626c2a7b09ce83d43a375f", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.1, 0.66, 0.24], "teacher_probs": [0.1, 0.66, 0.24], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.6373100280761719, 1.01171875, 0.66015625], "student_probs": [0.1013999655842781, 0.5274749994277954, 0.3711250126361847], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.1, 0.66, 0.24], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "4af15dda8a23f42c8546f886acee15878b85b8e9cac8a6ed38cb91d7b8b21905:action", "state_id": "2c5d8187793efed64254644c32068a0097a10d1cce3d11953438a224778cfec9", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.07, 0.6799999999999999, 0.25], "teacher_probs": [0.07, 0.6799999999999999, 0.25], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.638671875, 1.22265625, 0.619140625], "student_probs": [0.09132426232099533, 0.5874226689338684, 0.3212530016899109], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.07, 0.6799999999999999, 0.25], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "e08f947dff4128182c9cd2b55b2b519eb7770fe763dd0f161fb3035bb25e3c87:action", "state_id": "32570c76e20bc2a24a41ed4f4575417b01178a760578a75bf42f9d666954dca5", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.5599999999999999, 0.4], "teacher_probs": [0.04, 0.5599999999999999, 0.4], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.641937255859375, 1.16796875, 0.400390625], "student_probs": [0.10054613649845123, 0.6143240928649902, 0.2851298153400421], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.04, 0.5599999999999999, 0.4], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "db25f986b6d1849127d13960bac3536ba9ce3c87f23ee575f21ed4e47d1586c3:action", "state_id": "5d404fe24823be58cbad82381e566e3616beef991ba3bd8d79bdf621a18ffabb", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.59, 0.39], "teacher_probs": [0.02, 0.59, 0.39], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.9609375, 0.83203125, -0.21484375], "student_probs": [0.10969715565443039, 0.6589793562889099, 0.23132352530956268], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.02, 0.59, 0.39], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "6cc7792bc7d51635aea3be427ee85dbcb443d31fe6d6a31b9afa1a7bfca76cd1:action", "state_id": "48065dcf6d78a015574f022049466fe14161121c1bc9d9dffaa51746e6395e04", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.04, 0.61, 0.35], "teacher_probs": [0.04, 0.61, 0.35], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.978515625, 0.8984375, -0.3388671875], "student_probs": [0.1060514822602272, 0.6928945183753967, 0.2010539025068283], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.04, 0.61, 0.35], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d902c3f52b799f0ac86174b7d4bcb2e3060e98b1e8e8274adae875aaba33fce5:action", "state_id": "04e0e41d595be4cad15a12b316eab38843dc75a14b2a689219970e8593d45c64", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["east", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.02, 0.07, 0.91], "teacher_probs": [0.02, 0.07, 0.91], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.4521484375, -0.109375, 0.478515625], "student_probs": [0.20222273468971252, 0.2849014103412628, 0.5128758549690247], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.02, 0.07, 0.91], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "5265dcc276f8cf63314f06d1836bc18de9f549ecb79510a4aea2af9d94197ee9:action", "state_id": "fde3ac67d8a733efb730a2aafa87ead8cb9ba12973676018f4e20776df64b9e3", "family_id": "unified_snake", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["north", "south", "west"], "gold_index": null, "teacher_raw_probs": [0.01, 0.0, 0.99], "teacher_probs": [0.01, 0.0, 0.99], "teacher_rounding": {"probabilityDecimals": 2, "scoreDecimals": 2}, "teacher_target_kind": "rounded_proxy_distribution", "student_logits": [-0.543701171875, -1.67578125, 1.927734375], "student_probs": [0.07597748935222626, 0.024492256343364716, 0.8995302319526672], "gold_probs": null, "gold_distribution_probs": null, "gold_probs_kind": null, "gold_label_kind": "hard_gold_unspecified", "teacher_target_error": null, "task": "snake", "record_role": "policy", "continuation_policy_id": "c41dfc62ecc5a871dd9d9a56baadabc07fb58ad0996db975a9ca6e2aca767436", "training_target": [0.01, 0.0, 0.99], "target_objective": "teacher", "policy_target_kind": "api_policy_distribution", "scenario": null, "target_transform": "identity_rounded_proxy"} {"id": "d16f99dee73ef00422df263ff1e6996e10e19e234de49605dd43fb8203b9b474:action", "state_id": "29755cbdf7165310d807e89a3d822f0da9fa36a82cee896605cfabdf4345898f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3671875, -2.7109375, -3.17578125, 0.73828125], "student_probs": [0.6407153010368347, 0.01085320208221674, 0.006818342953920364, 0.34161314368247986], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5986e0a9989f971b3c212e721c25f590136049e099733598b9387654f3f73369:action", "state_id": "377b266e00a81127b5def44ac6ef3adb5f2e6a7216ebce7f55d4a343b852206a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3125, -2.6171875, -3.1953125, 0.87890625], "student_probs": [0.5956465005874634, 0.011704341508448124, 0.006565540563315153, 0.3860834836959839], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d5071d33c47b8cc89a95a494b6b2295654639bcbaa9c79c0c9eb663cbc47e106:action", "state_id": "65a936c7b6b08902c7c44fea41958dbaaaf074b0c366f00a9260844f5d0396dc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.421875, -2.5703125, -3.3203125, 0.59375], "student_probs": [0.6830384135246277, 0.012608404271304607, 0.005955788306891918, 0.29839739203453064], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "749906f8fb0b21c825d49f513b57e8cb8aea0a974c4e2f5aa2ba65a569d00c55:action", "state_id": "86700703a3b65737bdbae9f3caa0bcd3ccfae2dd473fc6965a4ddc54f53ebffd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.220703125, -2.08984375, -2.64453125, 0.390625], "student_probs": [0.6695818305015564, 0.024437200278043747, 0.014033102430403233, 0.2919478714466095], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "84eaa62c13b89d929a9c60f3d7954f1f777444e0802a5ad4761e6a71b821a564:action", "state_id": "15e9da7c6739351832ad42f621c2c88058ddd8352ff5ac43bb1336f73d8e6d6c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.1015625, -1.984375, -2.7109375, 0.109375], "student_probs": [0.6951469779014587, 0.03175930678844452, 0.015357797965407372, 0.2577359974384308], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c8b2fccd158156240e96fd62d21f8e1dccd7488123687b24772b403310d5c22b:action", "state_id": "f73ae226db8d850b3c5cbac9c72ec3296244d0366af821b407ff71104c751c86", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.05859375, -1.92578125, -2.5703125, 0.16015625], "student_probs": [0.6737083792686462, 0.03407017141580582, 0.017883725464344025, 0.2743377089500427], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cf628574a6f73a78962ef873ecad4041e0edfd9ee59f5bf22f3fd1b0c511b552:action", "state_id": "d932b968f1d1701218adda24036902659f86f30b5551f6534b14dee2e73b1ce8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.087890625, -1.75, -2.7109375, 0.3203125], "student_probs": [0.6472148895263672, 0.037893809378147125, 0.014495674520730972, 0.3003956079483032], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c75bb5cc5c172888cbf2b34a69a4459e4f62f735c426822d256770177abf3a7a:action", "state_id": "5f14246a0aab6e7e7a4aa2245678ef7cfa278d03d3469ceeb5591d2cba017a0a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.923828125, -1.2421875, -2.140625, 1.427734375], "student_probs": [0.35505226254463196, 0.040700867772102356, 0.01657361164689064, 0.5876733064651489], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ba33c018d6bcaf3a55bad63874a76b2c6a84e784e94c18405461dc56546b3070:action", "state_id": "0335104574f880a275b3441f687103fa53adfe5172f37ed65a13c9a83a2d8e42", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.0, -1.26953125, -2.05078125, 1.44921875], "student_probs": [0.3679487407207489, 0.03803141042590141, 0.01741204783320427, 0.5766078233718872], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b48405eb390a72dbcb81d6bed46de49132c54b32783a89ff894b814d127d9d4f:action", "state_id": "e365a8df009ab79e75657abf31c2961e77fb9c2983bf69cee07cfb6ef3d20091", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.453125, -2.1484375, -2.1875, 1.390625], "student_probs": [0.5017737746238708, 0.013688921928405762, 0.01316450722515583, 0.4713728427886963], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cc7b029e028905008b1bbb62cb76f82690ff50f9d47a4d38f003c13cfb7700c9:action", "state_id": "9f7dc7a1a327a0254e3ce21e955aebcf08a434500b41b1bac4bc4d0c47deb5f2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3515625, -2.31640625, -2.68359375, 1.1328125], "student_probs": [0.5414965748786926, 0.013823471032083035, 0.00957523938268423, 0.43510472774505615], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "28d2bca0f764f47f90ccdd9fa29f082bbe2eec1840b2292727fe6e283252ae91:action", "state_id": "09f460fd9d51ccc1394dd221e5d7f7fac47304d7b92784482b43d9388d95e20f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.640625, -2.3125, -2.87109375, 1.28515625], "student_probs": [0.5776944160461426, 0.011088628321886063, 0.006342838052660227, 0.40487414598464966], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "86e1854b6e57f60e8f672b9285baff0b851bb56100bc2e42f5bab0a22488b657:action", "state_id": "6d2cf967ecc1196005c8c0ba7e7b05d2c9a0be596ce5af669738b187e824901b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.5078125, -2.50390625, -3.23828125, 0.74609375], "student_probs": [0.6695003509521484, 0.012119465507566929, 0.005814983509480953, 0.3125651478767395], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0127d22bbd67f901f001959fcee7c8fa190fecc9f93e82359551f27ab53db7d7:action", "state_id": "343c4b8b1f9ca37512f02c3424aede486bcf09b03e6fa31f00fabb1c8f6fdd19", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.4296875, -2.19921875, -2.95703125, 0.7578125], "student_probs": [0.6452708840370178, 0.017128845676779747, 0.008028129115700722, 0.3295721113681793], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bf77600a151e41e3a4fb411cf70276f006c3eebb55070031c206870275e30bc5:action", "state_id": "d37f07bc9d3120c1cacf0fb4c1295715a3e03a65e87bd72f079b231d471a8988", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3828125, -2.28515625, -3.3046875, 0.39453125], "student_probs": [0.7107554078102112, 0.018144356086850166, 0.006545830983668566, 0.2645544409751892], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "900fa0da24a357208c77eefc18a6a4f74f4e72d64c52c06c3d609c7962c3944c:action", "state_id": "f74dd4a0c5007f5d061bd46cab754509534802f034921faab798005ce528f5b8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.2734375, -2.05859375, -2.91015625, 1.02734375], "student_probs": [0.5456094145774841, 0.019489428028464317, 0.008317066356539726, 0.4265841245651245], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9588ad4816bf48bcb0ef67b544d5760ee592e22412b064456e92b05dceb3e613:action", "state_id": "c90d39d1f43096461ab9cc46997cf4d3e6323f41d85a26d320e3e48559b7bd0a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.18359375, -1.99609375, -2.9296875, 0.83984375], "student_probs": [0.565912127494812, 0.02354118414223194, 0.009254940785467625, 0.40129178762435913], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a1da441a46b6e3b2e92b692356a0488e2d7be4815f7013285b1869447ef80129:action", "state_id": "cd27cfc6105a05646d89c091fcad2b3729a99170d67533c35bc7588e83195a8d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.060546875, -1.52734375, -2.3203125, 1.494140625], "student_probs": [0.37707552313804626, 0.028347952291369438, 0.01282743364572525, 0.5817490816116333], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3b43e30f38da210c773778901da3ecbb6a0222098684597f730174a5ffa80fd4:action", "state_id": "8122a4daf19a3b7f662618a84e0c0d23dea5896690358796ff277684cb4e0962", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.15625, -1.390625, -1.9921875, 1.630859375], "student_probs": [0.3664840757846832, 0.028705250471830368, 0.015729179605841637, 0.5890814661979675], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9b5099bb84850bc32609c5072bc3b07a75a6f40e65d1de0af04d46113bd28fce:action", "state_id": "28e182ad1c20c487f494666875206c327f42876d2aee22f633fe4d48bc054c50", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.431640625, -1.9296875, 3.201171875, 1.63671875], "student_probs": [0.12299499660730362, 0.004266592673957348, 0.7217471599578857, 0.15099123120307922], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5ddca5d274b6f54459adc5f8ce952893ce89533e9868b87e3a59d30bb430833b:action", "state_id": "36ac4b9302bef9d7dc7dc1a8d3bfb2016661e4c9245b926dc467cd32b3ecd57d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.36328125, -1.75390625, 3.1953125, 1.32421875], "student_probs": [0.1211749017238617, 0.005365810822695494, 0.7569265365600586, 0.11653275787830353], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2cb5f2d5295dd1f14796d38da18efaddb7bc1241325bbb5016be9f8205528dff:action", "state_id": "a148c0d07f07089a4724741796d33b78e3fc5915485112ea4914586c66db7e24", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.43359375, -1.640625, 3.22265625, 1.361328125], "student_probs": [0.12562261521816254, 0.005806996952742338, 0.7517056465148926, 0.11686468869447708], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bc1fc7e13623002594c8575ebb43a648e4d118a4ef69b6727c4bf019d52c7e3a:action", "state_id": "326a20727fb9a6219ab85cd9073d09df580fc4d54679ac533649b6062514e0b1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.123046875, -1.34765625, -2.05078125, 1.580078125], "student_probs": [0.36958372592926025, 0.031239213421940804, 0.015464532189071178, 0.5837124586105347], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3c6cd21898e9dd30b8071d9daf808dccd33f3883f5335c11488a7190d04de5f7:action", "state_id": "6cfdb892939f1810cedd2578a74d4abd690518ac1a3e1f1f4f81a37fce41d581", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.16796875, -0.9765625, -1.921875, 1.697265625], "student_probs": [0.34960511326789856, 0.04094677045941353, 0.015910200774669647, 0.5935379266738892], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a3b48ae21e60366e53373ad5bee1852b36da0e1f9f95fa55f2e66d337b774571:action", "state_id": "09a7e97032a19649ed89fd047708eec5c634e9bd925a9fe92eedc69332d46161", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.1875, -1.15625, -1.97265625, 1.55078125], "student_probs": [0.38813450932502747, 0.03724813833832741, 0.016464320942759514, 0.5581530332565308], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5451fc3896d56dac5d292f897f111ba877b03bc47217492a88fb60db0795953d:action", "state_id": "43d022d01217480376ffb4c02e220755368fb8ad42c71d81e28de35cda9f371e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.1484375, -1.09375, -2.03515625, 1.55078125], "student_probs": [0.37835970520973206, 0.04019159451127052, 0.015677891671657562, 0.5657708048820496], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "003194b79b982452291bd9de5d3306d161ef4217d9b1bf580a785e6483faea98:action", "state_id": "b019dea72493d44055431e2e595a5074c93d62dd299273566c6b35c1fe6e5aad", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.486328125, -1.90625, 3.203125, 1.68359375], "student_probs": [0.12790408730506897, 0.004300376400351524, 0.7119997143745422, 0.15579581260681152], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6ff6b45574493c426658fe244ec4210e48359afba4354f783f834dab6c4f672b:action", "state_id": "e2cc94f945ee51e8fc61918cf2a562b2d3f95dacb62751dba9e5bffe3196f267", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.373046875, -1.74609375, 3.197265625, 1.341796875], "student_probs": [0.12178222835063934, 0.005382181610912085, 0.7548002600669861, 0.11803538352251053], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e4380ab07e8e117512dffc5d88da913552df99df206f5f214e23e7a8a8d8cdfa:action", "state_id": "5a1992c8272521df705dc7505155f64989cc4b6e4c61da6a90e429f6d691084d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.580078125, -1.73828125, 3.23046875, 1.626953125], "student_probs": [0.13711370527744293, 0.004965187516063452, 0.7142271995544434, 0.1436939239501953], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8c25112bdacf90821f48b6d92705ee937139b59de8b957d473ab6dd05a4948cb:action", "state_id": "0a3bbee1c8602fd85bd361f565ec51d7a315a91c76bcf4c68f016bd0f41bd294", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.462890625, -1.6875, 3.234375, 1.470703125], "student_probs": [0.1260993778705597, 0.0054015167988836765, 0.7414107322692871, 0.12708839774131775], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c1bb9990eaf203d5c58d3a299fbb462961a63f95a1fc0d45ce067c749dcf7be1:action", "state_id": "cccb839725b4de7a9f6f47cbc3456fc8e72db6371c30526b3839ad3f9d14b1a4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.638671875, -1.5625, 3.109375, 1.876953125], "student_probs": [0.1501033902168274, 0.0061113787814974785, 0.6532940864562988, 0.19049111008644104], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dcbdc5e9086ff48776517f4159a251d7c205e7487c11ead89332cd5c4b33f9ba:action", "state_id": "be0354392b98a1c6c529352f9c7ec0c1891a1856164ca7387f25c5bff7d05046", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.14453125, -1.43359375, -2.06640625, 1.59375], "student_probs": [0.37267231941223145, 0.028291873633861542, 0.015025699511170387, 0.584010124206543], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "30c61429415b80bf3f83a5244815e0e560e4c8e0c795207ef947c98f7a57bc8b:action", "state_id": "35067e79aaf0e8b3ddf689a772f312805ea8cc5f53dbc3bb064e86436cd71523", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.244140625, -1.8515625, -2.01953125, 1.703125], "student_probs": [0.3750998377799988, 0.016970714554190636, 0.014346705749630928, 0.5935827493667603], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3b8eefa028f236ab02659f8d228389f4d3f9aed91707507c77531deb41f671a2:action", "state_id": "80f809e17a81e4d241b102147897153fd2590f251f773ee7647dd3c20f62dd66", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.1796875, -1.8359375, -2.140625, 1.62109375], "student_probs": [0.3787808418273926, 0.018566016107797623, 0.013689721003174782, 0.5889633893966675], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d2c7fc426db69cc132440eb6850e7448c4b488bf4890f9905379202bfbf9eb06:action", "state_id": "0d18cc39a0042a61c968e90f778534d7007dd70aaa952c1baa4651625ab8a95a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.013671875, -2.48828125, 3.578125, -0.140625], "student_probs": [0.026134079322218895, 0.0022003815975040197, 0.9486472606658936, 0.02301824279129505], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8713d4162e86715de92923477b583a83195d1433f0e8420126cefb7f09897880:action", "state_id": "41ca9eeba20a09422b3088a4c9dd465ea486842ff808c53e73a0c368849180de", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.07421875, -2.2578125, 3.4765625, -0.05859375], "student_probs": [0.031243031844496727, 0.0030336459167301655, 0.9383660554885864, 0.0273573137819767], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d9eeee8178f524dd3b0a17feac194bbeb565cc9fe882e94253767445a6b6b34e:action", "state_id": "62fc85b4ab6cbd77e101c283556a085c884bafa45573d86021377b1c9ba1976e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.2265625, -2.078125, 3.515625, 0.09765625], "student_probs": [0.034726377576589584, 0.0034653441980481148, 0.9312818646430969, 0.03052644059062004], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b0f0f6876510333fe4d8c5713db47f203a1e4ccc3a6ca3c920a7a33fc047ee5a:action", "state_id": "039bc85d50f82d5de9e78cb8e1fd73a9522faf9216bac5453aa65b0deec06d2f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.642578125, -1.71484375, 3.4921875, 0.673828125], "student_probs": [0.05152663588523865, 0.0048777153715491295, 0.8904333710670471, 0.053162265568971634], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ecbcbc5d2db8a50a1c8f43cc603cf38df0356ae7aa4e4e39d1f7741cd3fcc6de:action", "state_id": "1bf73e55c4ecbb37f8b9860a9c2d704c33490c7bb6208b6888ab1cd3e909a401", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.7109375, -1.552734375, 3.490234375, 0.7421875], "student_probs": [0.05481433495879173, 0.005698938388377428, 0.8829323649406433, 0.05655432865023613], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6c753bcda1420a36c02632022d4898d934b95e403e1c88101481e0bde93f8458:action", "state_id": "73029ccdbcf18fa081440847bc864f4f8a86143f437010851c7ad9bf829552dd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.53515625, -1.34765625, 3.232421875, 1.677734375], "student_probs": [0.1304083615541458, 0.007299882359802723, 0.7118992209434509, 0.150392547249794], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2a26375c76f5774f70ee3c72a1a5448bfe8403344c74eb29acf1b5f2ef52771e:action", "state_id": "021acb4d7b79c4b44ea2002f373c2a560ab1e3f6c8db4ca360237eab0271578d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.16796875, -1.08984375, -1.875, 1.73046875], "student_probs": [0.3439585864543915, 0.035970840603113174, 0.016404446214437485, 0.6036660671234131], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "80a75154a422fd6b32337f65f929976a6a60e9d3abdbf66b8723b89d570fa2f6:action", "state_id": "fe34977aa5d47e71d22745a7effcedc9cdb63539ac96c5a084c2536af3ee5ec5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.33203125, -2.265625, 3.5859375, 0.29296875], "student_probs": [0.035807106643915176, 0.002665762323886156, 0.9270917773246765, 0.03443535789847374], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e4e152dd7f2e3d3b45243e825c83defa744cf5a490f18ed9d051c9fbaea207b1:action", "state_id": "954503c95b4f4b42bdeab67b9efd8def1e394d0a1432dee825ef26bca6dc4a6f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.36328125, -2.05859375, 3.470703125, 0.328125], "student_probs": [0.040954191237688065, 0.003634891239926219, 0.9158715009689331, 0.039539411664009094], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bdf6ffd6a6768fd54d5e252318a5d49d5a1850ae307393f9bc587e19ca123001:action", "state_id": "eb83ec286c094ade0c8ee1654fe683d928361fa852c19db989cf451102ee5ade", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.345703125, -1.93359375, 3.509765625, 0.326171875], "student_probs": [0.038835618644952774, 0.003975064493715763, 0.9191049337387085, 0.038084469735622406], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f4e0d8e53ea278b22ef656123127a2327d6aaa52c4326defa6b82d2b683049c1:action", "state_id": "785ca33061a1be0185944ee726c19b7b87473ac1fd4b73e39a584bfb62a2de01", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.697265625, -1.693359375, 3.4609375, 0.712890625], "student_probs": [0.05566290766000748, 0.005097188055515289, 0.8827004432678223, 0.056539472192525864], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "631e1aafe9678572c3c3dddcf8054ef71484201549f5f2655bb118517dd5a5ae:action", "state_id": "c13a1d87d1a7e9610495b38b4b5ea103624cadd6de52202d8a4f8e8b8364b61d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.693359375, -1.56640625, 3.50390625, 0.7265625], "student_probs": [0.053313031792640686, 0.005564544815570116, 0.8860095143318176, 0.05511290580034256], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "331986e1369b8d4592ee9d225b20199be5cda44dae741d27ae18ad88fffaf545:action", "state_id": "5258f1f6a1ee5cccf6b2ecc74876af32e05dfefe892783e64203ebaaa3703f9f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.107421875, -1.22265625, -1.98046875, 1.62890625], "student_probs": [0.353680282831192, 0.034408897161483765, 0.016127126291394234, 0.5957837104797363], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2b5ef0b476fb2a7c496bd107bd4fcc466cd332ac553c11373857c7208a54ef5b:action", "state_id": "3ba956e0fc0aae8a01c008014d4c72ab3a252c8eb5661f78f0e9d29d538ed00b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.14453125, -1.06640625, -1.9140625, 1.681640625], "student_probs": [0.34872305393218994, 0.03821929916739464, 0.016373829916119576, 0.5966838598251343], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "020ecc4b80431dd84d637c8b2300a4f200c3a17641a6c5228eae440f36ff518c:action", "state_id": "c514855b4bebca713f3b679d79d9cb3e9bd0256bef74969bdd17497a9c67e00e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.5234375, -2.20703125, 3.25390625, 0.37109375], "student_probs": [0.05792414769530296, 0.003776001278311014, 0.8885607719421387, 0.049739059060811996], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0774b3f2226a41631d6f5b7adc5b79378158c8872dcd7cd12c1a2d212ee1b633:action", "state_id": "b40aa0126469882f5c66bc6b80828ed8e97e6095fe10200557284d9f3720c1c4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.47265625, -2.09375, 3.244140625, 0.34765625], "student_probs": [0.0557362399995327, 0.0042811608873307705, 0.8907955288887024, 0.049187056720256805], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6fc225db8e20b08c33727ad835fa49c740208954b05bbd1fe8f54e61d5bffb5e:action", "state_id": "b5eb857354ea31864518d4e75cd5efcc7b91c0986ea049d0208559a3af87b934", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.26953125, -2.03125, 3.484375, 0.16015625], "student_probs": [0.037180282175540924, 0.003724740818142891, 0.9257667660713196, 0.033328186720609665], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "de8f9a1131c08b0762433674abc49ecee0af1276223c68106f42d58fec57543d:action", "state_id": "4401808462ced42994bc95f01e437f9fe3ed40dce68e873b6eb5e2b4af95035d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.21484375, -1.982421875, 3.490234375, 0.17578125], "student_probs": [0.03505530208349228, 0.0038948734290897846, 0.9273374080657959, 0.03371235355734825], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5008b4d5bb9ccbc5ce38587c87549f68ae599e7432408e7e9044010661cc8c12:action", "state_id": "8f7a6c4095cb1bc71ff04ef6002fc93ee15e5fb396155f3113c5580386112b73", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.443359375, -1.767578125, 3.51953125, 0.345703125], "student_probs": [0.04220864921808243, 0.004625977482646704, 0.9148837924003601, 0.03828158229589462], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "86ff92bf46d55a9c049ef941f7256b97ae201a7d67a538488986d9c97dace7b4:action", "state_id": "25a4bc00d36886daa9189c151c5946519a85ccc9df73aecda0c1e45f66d02e44", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.541015625, -1.720703125, 3.54296875, 0.509765625], "student_probs": [0.04504867643117905, 0.004692778456956148, 0.9065958261489868, 0.043662674725055695], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bb442dd9f8c4fac95c1dd0df60f46b5be24a16477009aa8a2b8041e20836e95d:action", "state_id": "5b53427cf0ee127c03676925966ea0b7082a94ef4c15b6f9593feb758b1f1731", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.619140625, -1.662109375, 3.478515625, 0.611328125], "student_probs": [0.05116400122642517, 0.005226731766015291, 0.8928433656692505, 0.05076583847403526], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "53859808b2c6512ca8484ee908b4fcced0c2b656b0e8c5e377526d1a1f7ed980:action", "state_id": "dfea81457cd1ba6bbd350a09e639f03384f95bea33bd6a7433a0e25f08e2e3df", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.607421875, -1.05859375, 3.21484375, 1.865234375], "student_probs": [0.13598863780498505, 0.009455113671720028, 0.6785737872123718, 0.17598237097263336], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "73d75a3c4ead8368be63d1dc27d85869f38e1fd069d607daaeeb42baa741fa45:action", "state_id": "cded300eda31d3e89c6cab933befc4a3313655407e1b32c9e8d686274c0555d7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.060546875, -1.013671875, -1.96875, 1.708984375], "student_probs": [0.32398590445518494, 0.04071030765771866, 0.01566459611058235, 0.6196392774581909], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1384e87f23d616ee743dce61aaa7f1a2da20cc9c5b8e798e51a5020c2d816fc2:action", "state_id": "19fdce8af7c6aa8cca6ef26575144d8ac5d59bb328f527fc68c3dbe10785fd74", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.1953125, -1.890625, -2.0859375, 1.654296875], "student_probs": [0.37513113021850586, 0.017138686031103134, 0.014097897335886955, 0.5936322808265686], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b6f4ad73b0d460ba4fadb8fa8bef7631c2c6fd62d6c62333d9d179e1ddc9f0f6:action", "state_id": "93085388e93e7108ea8c1e2eec6c080a79edd9b966738a7fce2bda8c3790795a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.115234375, -1.953125, -2.28515625, 1.509765625], "student_probs": [0.3900846242904663, 0.018137911334633827, 0.013013315387070179, 0.5787640810012817], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cb49b29021edb34ae3b1b6602f39094378c7ec99ea7446e80e7c07661be9b6c7:action", "state_id": "04570377e588299d0183be310f8bf0144415d206934d8bd887f63fcd4bda266a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.29296875, -2.25, 3.587890625, 0.26953125], "student_probs": [0.03444620594382286, 0.00270859501324594, 0.9291969537734985, 0.033648259937763214], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fc60a51ac7fbb776c759c088429a712f0779c619eec106caa4109bb1b4c7e5b4:action", "state_id": "79b12947fa03fb798191c2195ac78cfcc1133787f7dbb4427b4ca7f77899df4c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.546875, -2.12890625, 3.37109375, 0.6171875], "student_probs": [0.052660755813121796, 0.003625851823017001, 0.8872166872024536, 0.056496743112802505], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5863813f3e22864cfcb6d11755b8a8c3ae6135fbbee4dc179b94c460c0f4c335:action", "state_id": "d8f7a53524ac403e853b3e914c94526293ad421c7460a6fc206be7bc37b72506", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.384765625, -1.8984375, 3.35546875, 0.21484375], "student_probs": [0.04661718010902405, 0.004752952139824629, 0.9092974066734314, 0.03933234512805939], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8af160cf72dc39aa6c6acfdaf9423e5dbb7208d620672fe9d0ced4ca27aaffd4:action", "state_id": "35bd07607ba1dfcebf62fa12fb0c21b3ecac475e976620d9923165d7af3237a9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.349609375, -1.7890625, 2.873046875, 0.2734375], "student_probs": [0.06889016181230545, 0.008116032928228378, 0.8591563105583191, 0.06383754312992096], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c327a717e789d73a6f3affd7bae773149ad3a9e4b118ae634e970e45bca5c8b3:action", "state_id": "6bb7a844568956089b5cdc167c5d896ba82fa0bf206bf77a9c48c185811ddb82", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.88671875, -1.42578125, 2.43359375, 0.783203125], "student_probs": [0.1493106633424759, 0.014783757738769054, 0.7012777924537659, 0.13462774455547333], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5854c0a09cb9c7993f320e1be8ab218c11d8a0f8e385242e56e34584db2280ea:action", "state_id": "d1b34e2d96e58cd9592ee2b26b4857cefc8bda83e72e85ece10306cc069ea572", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.302734375, -1.23046875, 2.078125, 1.380859375], "student_probs": [0.23083481192588806, 0.01832927018404007, 0.5012440085411072, 0.24959194660186768], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fe02307463a764792550d5b0787aa33f6950f288a102a84f295ee1fde1593e18:action", "state_id": "09804a43da4c1f2a4d35294989961e7e610c83364a9a826f8b0e30e4040dccf2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.02734375, -1.08203125, -1.72265625, 1.583984375], "student_probs": [0.34129196405410767, 0.04140341281890869, 0.021818064153194427, 0.5954866409301758], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4214c20927c1d951aec75f28195b0e344607930fd0799860f1d29823350241a4:action", "state_id": "2f81a150621b34b6a4d87573c1e69f095e8468a33cbcda78fbacf17eb8fb8496", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.08984375, -0.953125, -1.859375, 1.587890625], "student_probs": [0.3536657989025116, 0.04585038870573044, 0.01852523162961006, 0.5819585919380188], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b73061aadcbefa7bb341db5f6f46cd0f57ebdad8b1ad284859131b208705980f:action", "state_id": "e70603202fb711eabc52b9e24003449f9a243098bc17cffd9f43df25e24aba18", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.046875, -2.50390625, 3.578125, -0.1875], "student_probs": [0.025329776108264923, 0.0021704824175685644, 0.9504928588867188, 0.022006891667842865], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "84c291fe5c1fd85ac4b12709a9a9014bb0115cfa582c4e2087f2d79c5736b039:action", "state_id": "09b3d1b29dd4008f058f481dfb37df7edc9ac7ea22f32cbbf79e6d19681228d5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0625, -2.296875, 3.458984375, -0.068359375], "student_probs": [0.0314161516726017, 0.0029681746382266283, 0.9380530714988708, 0.027562681585550308], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5d872b6c2d687fba67e647f52893a691db5de58cc22132cdbabdaaebaa885bd6:action", "state_id": "d6b824c817f8a8c88e0dfb574aa0936cb4d10d387e517585094bc1f7f24352f9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.08984375, -2.15234375, 3.49609375, 0.01953125], "student_probs": [0.03106526844203472, 0.003299935720860958, 0.9366788268089294, 0.028956016525626183], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "284e8d602da80c4fa861246702f63926e3922cb1716e2587641f75901702d134:action", "state_id": "5388cae9b62add7392595a8e95d8644aaee2c78b6965ba6ea445d04c7112c594", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3359375, -2.00390625, 3.3671875, 0.150390625], "student_probs": [0.044149886816740036, 0.004253518767654896, 0.9149234294891357, 0.03667309880256653], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dd18526bd655bd29cdc4a36bf814ebca6856ee8d0a03ab58795ba7f5d6618e14:action", "state_id": "c7ff7bd3361629ea01f50b35c2cf1d5c9774e6054f7a7fe3c06396ac1b62c125", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.46484375, -1.6875, 2.87890625, 0.46484375], "student_probs": [0.07521343976259232, 0.008740664459764957, 0.8408324718475342, 0.07521343976259232], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "352c28d9409efac128709e67e919320020eb67e7de89833df7cff2ab2f464b60:action", "state_id": "ddf8cf8452f0fb266cd50cf4d73560fbf642b049bce264d60bb756658756057a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.64453125, -1.51171875, 3.1328125, 0.486328125], "student_probs": [0.07137759029865265, 0.008262556046247482, 0.8594264388084412, 0.06093335896730423], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fffd5abfe7871a6e10a43f5be50e930df9a76ea3e7ab04b87af6bb283d8fc0bd:action", "state_id": "513dee3544c077286b11d57ea40f165d332db77ac08d7d7f963ebde11a887054", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.810546875, -1.46875, 3.150390625, 0.65234375], "student_probs": [0.08106587827205658, 0.008297591470181942, 0.8414325714111328, 0.06920402497053146], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f147cc9f4bdb19233e90d38fb093586deb837cc2f65fe2a0df6c3d85434cd847:action", "state_id": "39a00ced08dc668e3fdc9a4f3f010a2e5833a28bd094f6fa3b6abc21a6505c35", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.591796875, -1.009765625, 2.359375, 1.751953125], "student_probs": [0.2271491438150406, 0.016844840720295906, 0.4894023537635803, 0.26660364866256714], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "797b0cd14da6772841e8f524f172de77b6c04e462b5afc313977aafcbaa507a9:action", "state_id": "dd1194665d4f798407897b87e9e8c1835a618820520a95d251cc8c7156b30345", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.029296875, -0.984375, -1.7890625, 1.625], "student_probs": [0.33250126242637634, 0.04438811540603638, 0.019851593300700188, 0.6032590866088867], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "06eca35fbfffa53b56ee677cfb87027e219540e3c2b4fc1b0f69f5e917877c16:action", "state_id": "c045f484559deffe91cdf38dc35411987be8866a8bb40e228cb1843ee4454caa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.03125, -2.51171875, 3.578125, -0.166015625], "student_probs": [0.02570655755698681, 0.0021517411805689335, 0.9496762156486511, 0.02246549353003502], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5298bac1c43e9c61d5212733a3c7a5176b0984a2defad77b9bc7f749579328f6:action", "state_id": "df393c104241fcd34c588ecf8e6d6e8f7c726122eff7e9301ac878c273d9dbdd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0859375, -2.2578125, 3.4765625, -0.0234375], "student_probs": [0.03156878426671028, 0.003029564395546913, 0.9371035695075989, 0.02829807624220848], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "639f69756f004f70d8506d5464ddb9c841f36f21f415e159712b7079493b43b1:action", "state_id": "d1e88b1729161b5edbfe9b10314c5102d134f91fa6ac1c0682db3ff4ffc61739", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.09765625, -2.140625, 3.513671875, 0.0078125], "student_probs": [0.030799012631177902, 0.003284457139670849, 0.9377639293670654, 0.028152575716376305], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "12fe72738e248824e13a4efa0a44b5e43cc55ad819c1269a7025113b84c4c0d6:action", "state_id": "d62cdc60c90b92eca5a6237863dbc4f15eb9770663d878f95745385970ec9cdb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.392578125, -1.91796875, 3.478515625, 0.248046875], "student_probs": [0.0419241301715374, 0.004159166477620602, 0.9176343679428101, 0.03628230839967728], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "df10c51d7ff98ddd3f475e1b8d047cc6f855db2afe33e06047222edeb58e231c:action", "state_id": "e6e2f25d569be03597aca4df7aaee346d4ea1c0a605225c073e69ffdcc370f63", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.654296875, -1.60546875, 3.517578125, 0.677734375], "student_probs": [0.050898343324661255, 0.005312511697411537, 0.89168381690979, 0.052105359733104706], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3e94ec61e9be8ea0a3309b725139cdf1b270ed1678d2455f46f9cdcdc9c36444:action", "state_id": "ea166b0a24865b9500ef5ad5a88cdefdb9ae1f55b7a97ba7e65e53d3dac0e65a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.626953125, -1.697265625, 3.51171875, 0.587890625], "student_probs": [0.05010290443897247, 0.004903063643723726, 0.8968105316162109, 0.04818349331617355], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cef40435e13b300c25fdc1467712e94dcc052f6334651cc1a9bbef309f5a43ba:action", "state_id": "6f56f41f45af30c29085c824596dba60fb518f2c73faa017a99840bdfcd83073", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.603515625, -1.14453125, 3.130859375, 1.828125], "student_probs": [0.14447082579135895, 0.00925376731902361, 0.6654219627380371, 0.1808534413576126], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "86d0e72725fc2844e3e17539d81a4ded113ab025cef552e50078c918f36b1e4d:action", "state_id": "cb8aa09845d7e678585978e62f0638c172b44c95874d92b12b70cfab2cf737a6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.15234375, -0.974609375, -1.86328125, 1.759765625], "student_probs": [0.33289971947669983, 0.03968162462115288, 0.01631714403629303, 0.6111015677452087], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f741da7ec1792e4d5ab8e14f65d6984751af70c133c6504b3c9d4658bdd418d5:action", "state_id": "fbedaf26897a35c7b3e5f88698cee5222724a4a70f95562038e43d62688e625f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3671875, -2.71875, -3.2421875, 0.7109375], "student_probs": [0.6470153331756592, 0.010874629952013493, 0.006443004589527845, 0.33566713333129883], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cf38a8ebd05a296e180303e6c9e72b9940463c45fb4326e81c36d83941f498ec:action", "state_id": "3e1acec01a35fad5456424a6e24d05ce1201c203e3c1bcfc247c9f95705ec059", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.27734375, -2.61328125, -3.23046875, 0.8671875], "student_probs": [0.5899698734283447, 0.01205460261553526, 0.006502969656139612, 0.3914724886417389], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d72d08b3062a0288da6c7d8a3eff40f2b25675116167f4f22cf3a6913532cddb:action", "state_id": "578c4b567820994029de250658a41a0c77c01323236ffc400d22b19c661f7813", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.58203125, -2.578125, -3.3671875, 0.64453125], "student_probs": [0.7070637345314026, 0.01103381346911192, 0.005012336187064648, 0.2768901288509369], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "84922c91c7ee8e933e5cfc48b4085f26315f7ee0e426853b2be1b3ac1ff29483:action", "state_id": "cc4c6cffd31c57670342d596b47b89f08ec54ef709a348dafe46eedd300a7cd1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.5390625, -2.58203125, -3.33203125, 0.734375], "student_probs": [0.6797544956207275, 0.011030211113393307, 0.005210302770137787, 0.3040049970149994], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "770b49f75caf00b0bf6e3ffe3a2a55f45f68ed3612f274e837d8bd566364d8cc:action", "state_id": "1091b87998b65d54506f1ef60f753f7b202f8b80628bc299e12cae0b13ed46c4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.38671875, -2.16796875, -2.98828125, 0.7578125], "student_probs": [0.6351815462112427, 0.018160035833716393, 0.007995755411684513, 0.33866268396377563], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a54d454631523731eed7701d969d85cc33b083c8be9fcf428d2a064d2a9df386:action", "state_id": "cea7fc8679bb6b944e0dc86b2f3048a5fc5924870bb359f6e11942e0b9e8c7a5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.21875, -1.515625, -2.24609375, 1.572265625], "student_probs": [0.39678165316581726, 0.02576485089957714, 0.012410493567585945, 0.5650429725646973], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "08b7c3512efa4f2402d000d9cbeb8ab03e8e3e1f34d5b19c0a18031637306c3b:action", "state_id": "ad01fe34e2497d09eb0a5d875437896fa53497f37fbc9c075bd02c653a95acbc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.080078125, -1.4296875, -2.15234375, 1.5625], "student_probs": [0.36486467719078064, 0.029658861458301544, 0.014398221857845783, 0.5910782814025879], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0772eb78fc545014ccd12fa0937f96b13f2e36481f06cfb2aaae65dcde40f572:action", "state_id": "479b27ef1f2a0b0961f5723dd306e68ccb69c30a732c62fdeda8b390baa4b182", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3828125, -2.234375, 3.525390625, 0.328125], "student_probs": [0.03970886021852493, 0.00289906095713377, 0.9197965264320374, 0.03759559616446495], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "74e149350e45d8a0c3ad56d170c0851c9a332e09d056e5a17acc4692c77056a6:action", "state_id": "9288d02d703d39722e06b47348b79760cdef037921dae60f1bc55a02cc22e1a1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.4140625, -1.98046875, 3.45703125, 0.41015625], "student_probs": [0.04337507486343384, 0.0039564757607877254, 0.9094624519348145, 0.043205972760915756], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "12a27581a57cf313c61779b0113819921c675ea3c8555b37206f821005d195cb:action", "state_id": "35512aa9b306522dabb2f6389d02d4af430a2fb3d28997b1f622aa40abbd7b7e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.52734375, -1.822265625, 3.515625, 0.50390625], "student_probs": [0.045612581074237823, 0.004351732786744833, 0.9054796695709229, 0.044555965811014175], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3f2021782c5762cc11e0eedca1a09d95da2f8bf5b76dd2bee706bc3f4f77e0f8:action", "state_id": "58b53bae8b7a0ac1a62f461140e78aa111dafb190b02245884aad6029db040e7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.482421875, -1.42578125, 3.130859375, 1.59375], "student_probs": [0.13566339015960693, 0.007403654046356678, 0.7052936553955078, 0.15163934230804443], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "231ba1b4e13c9489c5523556eac6d0e04ac1b88b620beed14a363508850784a7:action", "state_id": "db428090da3b17b276a25d1930548ee5db44a101f84c6881df1aea89edc4a7a5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3125, -1.21484375, 2.8203125, 1.408203125], "student_probs": [0.1493174135684967, 0.011926115490496159, 0.6744427680969238, 0.16431370377540588], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7d4b56f3d5af5e5fb04dfbe769ba869ce959042edc3a628ebfe02a89c3d53bc5:action", "state_id": "d29d7623b1659f33e151f6ae95b31bba2a63d3c1d5d222d08ffedeff38289221", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.029296875, -1.2421875, -1.59765625, 1.140625], "student_probs": [0.43606826663017273, 0.04498433694243431, 0.031527046114206314, 0.48742038011550903], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c46dc8c94e59263810bbe64ea30702a3891139f7ea262fb69aac22887a83cbe7:action", "state_id": "4b217c420309a55c2f702353809b0be8501fa708cefbe63c014abb5e9b2d0dee", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.919921875, -1.234375, -1.765625, 1.158203125], "student_probs": [0.40762507915496826, 0.047278277575969696, 0.027793465182185173, 0.5173031687736511], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "11977eb5564d251533ffab9c351c5aed99d155e0ed1ab7c7ae932d376279f712:action", "state_id": "d00d4908e371d85cb1269c39fa227f47ec3133f2db1e550adba8ecc2d730aef3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.9609375, -1.06640625, -1.9296875, 1.34765625], "student_probs": [0.3760291039943695, 0.04951733350753784, 0.02088521420955658, 0.5535683631896973], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b0afdcf653b2b2ef0aca826062b050725998623993f15fb4426b39d50c7094d4:action", "state_id": "913add7039aae993a9a7cbe271e14bafde42153d7b1693966193d1bd33138dcb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.37890625, -2.44921875, -3.0390625, 0.90625], "student_probs": [0.6034444570541382, 0.013125134631991386, 0.007276756688952446, 0.3761536777019501], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "28b0e21e2f9abee11a4c5afbb46a88259f7910d575b532b83ef3c70ab084d6cd:action", "state_id": "900af2d4aada801c637e020f167d3ad8305b718f8b0cdc68cc59d1d0843dd461", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.28125, -2.40234375, -3.0859375, 0.90234375], "student_probs": [0.5805754065513611, 0.014591306447982788, 0.007365685887634754, 0.39746758341789246], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "12adf5388ca852d01e97cd5463034a56fe14dfcc08ee90e3703ec6cfe9936504:action", "state_id": "4c15935c2840ed4b3708d2ee7a1c30d387395d9e70a71af49c95af1ca2bf9b84", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.57421875, -2.43359375, -3.06640625, 0.91015625], "student_probs": [0.6482643485069275, 0.011780976317822933, 0.006256829481571913, 0.3336979150772095], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9a7acbf26dd7072a685df8210f6e18ea3e8e05c361dc43e832e633d4006cbed4:action", "state_id": "0031eeab567f899a98560bec555e9d3f7bbc552282139a4f8b104b980e870414", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.37109375, -1.87109375, -2.4609375, 1.646484375], "student_probs": [0.4205588102340698, 0.01643473096191883, 0.009111642837524414, 0.5538948774337769], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1bcd9dbbcf3774f22baac45a106bd5dc9abe3d4f02815f54a5d59397142829cf:action", "state_id": "b3a7c816dd4ce30df167487788dd193cf726c82909684e4f21dba11b43be831e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.26171875, -1.4921875, -2.1484375, 1.720703125], "student_probs": [0.373248428106308, 0.023767950013279915, 0.012330650351941586, 0.5906529426574707], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a888c064e4e9a8a2d857646b38d73b76af38ecd9c94a3c7a65feee6b974cde69:action", "state_id": "f116b31d2092f695e6af53e902cfff0651119dfb6f86b1375cf894cc87462bcd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.283203125, -1.5, -2.19140625, 1.560546875], "student_probs": [0.41451913118362427, 0.025633905082941055, 0.012839286588132381, 0.54700767993927], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "edc68dd2c0e9d289865f41934c21a9ae9ed842316644b5f53dd5dae53365f7e1:action", "state_id": "ed21114e18173cbb591c5cb64179e54f181c88719728696b5c72c14b264ed033", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.208984375, -1.3984375, -2.09765625, 1.595703125], "student_probs": [0.3872207999229431, 0.028547609224915504, 0.01418740302324295, 0.5700441598892212], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ff980899a90e31a1db2cdcd93f68b9417d1bc58d2834ddce988580f68f8c1cee:action", "state_id": "e5cb3e5e8e5e81a939d54de9ccb31789d60c69d6a1a523f8c06f138071080135", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3984375, -1.9375, 3.201171875, 1.587890625], "student_probs": [0.12033142149448395, 0.00428153807297349, 0.7299559116363525, 0.14543117582798004], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f519fa5b086d110ff0db2d7a5b4d317d1abc73964654575e49065e564c39e9f3:action", "state_id": "e99bbaefb20ff130a1731843ad11ab006721f3c11d578b49eca2d48a1259d485", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.33984375, -1.73828125, 3.1953125, 1.26171875], "student_probs": [0.11953730136156082, 0.005504155997186899, 0.7644045948982239, 0.11055393517017365], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "172570e12838d3c281544ad00a69725c0461d1014387f3dd07ab0e0abbb08eeb:action", "state_id": "0efdb2c53796128af14120f4d329e682d14275ab16d9041dac9f97329da66f2a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.43359375, -1.625, 3.224609375, 1.361328125], "student_probs": [0.12542682886123657, 0.005889249965548515, 0.7520013451576233, 0.11668252944946289], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "578af901d57bc2a593791c661d0f4c32cc361ada2e40c76147b721daa1056194:action", "state_id": "abe040e7595ac6cb1a7cb00fd9a94953cefc4e0a4623d9726d4fb55992f0b0b3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.53125, -1.43359375, 3.193359375, 1.59375], "student_probs": [0.1353824883699417, 0.006981475278735161, 0.7135220766067505, 0.1441139131784439], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cb693ae0e3c903d36d3dd67cc0b5861ce07c9c2ae64a3f08a46b5984e2d409d4:action", "state_id": "87d5217051db75bcc8957737e31736b1375c9f5ce9fe7720a2a1ac0e0074769b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.162109375, -1.0078125, -1.87890625, 1.748046875], "student_probs": [0.3379923105239868, 0.03859417140483856, 0.016151413321495056, 0.6072621941566467], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "db9e3ec63daab1b5254672d3aeed55b949f000c4a22e14809e9aca02e68263d1:action", "state_id": "88972aa21de650020710b12337b3e715f99ce38ac7048ccfabd23aff785b3392", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.47265625, -2.20703125, 3.26953125, 0.33984375], "student_probs": [0.05453290790319443, 0.003740116488188505, 0.8939763307571411, 0.04775060713291168], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "235ce86972cad3b7090cce4047e3bcdf917945bcb156c77c76d18dcd61a95a39:action", "state_id": "81a95b29b4c9e757b57876abddca8d9c4938c0dd2cbbe5f8161e6c2325897191", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.51953125, -2.03125, 3.23046875, 0.40234375], "student_probs": [0.058786142617464066, 0.004586535040289164, 0.8843417763710022, 0.05228547379374504], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3a78b89899c549d543c54b58111fc6acea35f4358c5a028d76c827c87b62f024:action", "state_id": "d4b5ae14e3bb98be0de2a7c11d82ccd50356919484928a556dd91cc88dc5bc67", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.43359375, -2.005859375, 3.501953125, 0.21875], "student_probs": [0.042734190821647644, 0.0037267860025167465, 0.9190667271614075, 0.034472282975912094], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f42afee3da8fcb618b84ea22350ed2a053f52aeda00748931972cbbba76913d9:action", "state_id": "7b2c324aafc27355155453bb4711c358bb3fa5ca73a3bb2bd804c05f36939042", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.24609375, -1.951171875, 3.490234375, 0.208984375], "student_probs": [0.03608240187168121, 0.004008991178125143, 0.9251407384872437, 0.03476794436573982], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7473a0c9d27b6fd73eacdce72240fb69d915262ff041573f12211355793239b5:action", "state_id": "fe5ac2c013bbd97d7df4982f8fe5e222d09406c9d2e56ed49e5ed4048ee6fccb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.443359375, -1.751953125, 3.505859375, 0.369140625], "student_probs": [0.04269720986485481, 0.004753214307129383, 0.9129065871238708, 0.03964301571249962], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "eee165553d28f2f7a8bac1d933675c19665a8fde1cd15b9ac4abf5cecdb8986b:action", "state_id": "615734be029853fa5e7f57ffbb83bef5a8d1d19536e1549188341464e2cbc1e2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.572265625, -1.720703125, 3.525390625, 0.525390625], "student_probs": [0.047123298048973083, 0.0047578634694218636, 0.9031534790992737, 0.044965364038944244], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "41ea2a43763c7d524e94579406ea8ee60d34411d2cfbcc516c304d45787ceeca:action", "state_id": "0cab6890d050220854382237afc86a88390ce0e6ebbe0242944557e7264229e5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.642578125, -1.67578125, 3.478515625, 0.626953125], "student_probs": [0.05227581039071083, 0.005145766772329807, 0.8911129832267761, 0.05146534740924835], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9a65f42cc2563eb6230ab66c4383bb7a6d3508c62db0b935ac18507d8a4a2b01:action", "state_id": "ca0da50cdc245b6db85fc34b8dcfb56ddcaa27e936f1cddab16021f64b7058d5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.609375, -1.05078125, 3.19921875, 1.8671875], "student_probs": [0.13760806620121002, 0.009623934514820576, 0.6746899485588074, 0.1780780553817749], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ec5cab1a84d45af039624b82435fc31e8417aa39f5357e1d41ac116e724b0222:action", "state_id": "abfbf8a55c1a11dd8cb783e57da31d49df93c1ac4c4ea17610fede5fe7dd5316", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.0859375, -1.013671875, -1.9296875, 1.7265625], "student_probs": [0.3258192837238312, 0.03991425037384033, 0.01597009412944317, 0.6182963252067566], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c92fbec849fab6075cc8e10b26f40e9d20c3496b75c6f6aa48a96fc8fb623be6:action", "state_id": "f54ab505c45f5a5b084a985ba5299cf189e43b07e83756baa490b314011289fa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3671875, -2.72265625, -3.2421875, 0.71484375], "student_probs": [0.6461937427520752, 0.010818478651344776, 0.006434822920709848, 0.3365529775619507], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7c468ca8fe09ece49a52cfe3f5219a1126e3844649f093b8dccd28acaf9032d3:action", "state_id": "569a223901fc6cce34fdeac536c3604b9916ddf1e79708837ffe781de721f92e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.26171875, -2.63671875, -3.390625, 0.453125], "student_probs": [0.6778296232223511, 0.013742022216320038, 0.006465964950621128, 0.30196231603622437], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "499cac50624990204bb3119b021d3896e146637da3d92468bb273abb5960393f:action", "state_id": "d62a5c2db6e3aa189adeec2c1b70ba6db3f276b6d8e604e1009b4c5c6ac756e2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.44140625, -2.46484375, -3.03515625, 0.859375], "student_probs": [0.6288317441940308, 0.012649450451135635, 0.007151350844651461, 0.35136744379997253], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d3f288f250ec73f346e97465fad00e21afc6176af0aa21646d7365be1641e130:action", "state_id": "68c05a1a8b17ca13ee7ccdc893d07c997e82b422bce5d203d418cef98ffa8d1e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.5390625, -2.578125, -3.19140625, 0.92578125], "student_probs": [0.6382910013198853, 0.01039793062955141, 0.005631216801702976, 0.3456798195838928], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c100d9f2d03a71d0204e92dc6dd9875e4b3a66ab07010a76f203f4919d8ac582:action", "state_id": "b535db41f9b04d260ec0e4f35001e8c45de2c826b3713714765effca01ee3e2e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.54296875, -2.41796875, -3.03125, 0.9609375], "student_probs": [0.6296746134757996, 0.011992311105132103, 0.006494687404483557, 0.3518384099006653], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6a1b5ea98d91281844cf00a7385783013dbe9b4dccea019090c94efec4c06947:action", "state_id": "0272da34ebd9c091b0c01fad330ecdbf674a92adb003b33344cbb3d49d0ba890", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.328125, -2.31640625, -3.2578125, 0.3515625], "student_probs": [0.7077484726905823, 0.018496055155992508, 0.007214921060949564, 0.26654052734375], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4f122109fb5df5e665fd5d605a0f312486582a39872a40a3b5d06699d5cde967:action", "state_id": "53345d424dc33444856ec01fa3ad6fa37866d6eda2f9b7a362300e883e927413", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.107421875, -1.5, -2.26953125, 1.62109375], "student_probs": [0.3598037362098694, 0.026526302099227905, 0.012287784367799759, 0.6013821363449097], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b63601549e6a8e01cedf77db34c74357c12dcf0371bd9bc52691d3a259b2b1f1:action", "state_id": "41307f99370c92d4bd41a89a1839c450846c4e3ed29f6f879babf8a32d441f34", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.08984375, -1.3515625, -2.09765625, 1.578125], "student_probs": [0.36260440945625305, 0.031560495495796204, 0.014966472052037716, 0.5908686518669128], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "08f5e5ca673a4427220ebbb5d435a8ac189ef6653cfda318bb313701e68c9428:action", "state_id": "84612cb2c0ca371373ba2005fa1c3ec2383eb671b6cba632c5878106d71b1270", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.5234375, -2.20703125, 3.25390625, 0.37109375], "student_probs": [0.05792414769530296, 0.003776001278311014, 0.8885607719421387, 0.049739059060811996], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "35405795abf80599743f987857a8f73e29217c2cc64c07e4fe9d9994efdfc367:action", "state_id": "1c3280bd035fcc4d9d9808dd9240cfd690fb6082dd62ba76a1e2d7b94b3b351f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.47265625, -2.09375, 3.244140625, 0.34765625], "student_probs": [0.0557362399995327, 0.0042811608873307705, 0.8907955288887024, 0.049187056720256805], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f855ae31148858cf2f1980acfec1baa3e2525826c5093becea7bdf12fe734634:action", "state_id": "733f9b5a44bab6835f59819f463a1cf066ca4c830ba5f6afaa7ef59faace2f63", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.26953125, -2.03125, 3.484375, 0.16015625], "student_probs": [0.037180282175540924, 0.003724740818142891, 0.9257667660713196, 0.033328186720609665], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ad7488129b06f1039050f2a0b9b9dc8451d275627e95dd0380c6cf57c1995583:action", "state_id": "66d95a6b63665238f120827a5480ddbe6e5b957da944f0114efa32dfc7dd1fe3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.2109375, -1.966796875, 3.490234375, 0.19140625], "student_probs": [0.03490273654460907, 0.003954408224672079, 0.9269152283668518, 0.034227658063173294], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fac41c4d5d988dbeb763becc18edd083ead722bce606dfd070da8cddb92e24e9:action", "state_id": "a27df0055bb762b3f64daf6230cc1509bafb9beafc01d5b6713ae8eba579f5a7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.455078125, -1.767578125, 3.51953125, 0.369140625], "student_probs": [0.04264625906944275, 0.004619485232979059, 0.9135998487472534, 0.03913440927863121], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c03878b8e605f18ff49340ccf71244088707fa3841c0ca8432b0a8c34d4999ca:action", "state_id": "89b9d2a00ac60f1b99d6f3d7c128fcd9bc9442e749ba2594fbab350d0806ae1a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.541015625, -1.728515625, 3.525390625, 0.505859375], "student_probs": [0.04578135162591934, 0.004731988534331322, 0.9052868485450745, 0.04419981315732002], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c802fb7f92e5a93f7f19a27014b80546cceacc7656580b5724184cef779323cd:action", "state_id": "e43298634359d43b35b043d45b745e1062de1dd1b94c8039f190b1cf04a0a2bd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.619140625, -1.669921875, 3.478515625, 0.611328125], "student_probs": [0.05116607993841171, 0.005186268128454685, 0.8928796648979187, 0.05076790228486061], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "baf4b1b97408eb750c1a4a7477577677cad1c9a7530e89e0c52c24e402bebf4a:action", "state_id": "016f9a7f9db1ee2ee3e3b4f0116674b965d16439a7430d230ba25d748fb0d41d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.607421875, -1.05859375, 3.21484375, 1.865234375], "student_probs": [0.13598863780498505, 0.009455113671720028, 0.6785737872123718, 0.17598237097263336], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6f782fce3d6c03b7759a0a1aeecaa4093bbd428dfb881d6a9762de520a799596:action", "state_id": "7773223a596bddde217dec33309b5c9dad97ad77fbba6f19cf5054c8fd652557", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.060546875, -1.013671875, -1.96875, 1.708984375], "student_probs": [0.32398590445518494, 0.04071030765771866, 0.01566459611058235, 0.6196392774581909], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cfaf283d123aede98fa5bb555f43a54c4aa6bef1d12a0f030228b5e810252929:action", "state_id": "ffb666a261d695d4d670a14faff2d68b8d6baf541500ec502fc87c3f9188a162", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.015625, -2.50390625, 3.578125, -0.173828125], "student_probs": [0.02610493078827858, 0.0021680821664631367, 0.9494417905807495, 0.022285163402557373], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5bedf55880428db87db15e072b9b381885ea44591dbe1b536221ea35e4a300b9:action", "state_id": "921e9089bc5124e0a16f289406fb5b8c7af07dc25157be0843c9dbcc06e97596", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.07421875, -2.265625, 3.4921875, -0.060546875], "student_probs": [0.030790405347943306, 0.002966430503875017, 0.9393347501754761, 0.026908373460173607], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0c76c3e3328a633a4377cdf73819928cebaf63f2e76b4ee317fb3c5ab6e84b86:action", "state_id": "c19e9c33925bbdc97155c8cb50e351fd9da09dff0308fab08ebbb58e513828ca", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.20703125, -2.078125, 3.515625, 0.08984375], "student_probs": [0.034085698425769806, 0.0034684978891164064, 0.9321293830871582, 0.030316447839140892], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ad481d828fb80aba7a096b6da8490228f7ec948e6424ccb9f599850e64e017ee:action", "state_id": "806ddd8bb5ee8dedd1b36cfd1965fc7e7710f0464c6a4f39643e255adc89c21b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.626953125, -1.70703125, 3.478515625, 0.642578125], "student_probs": [0.05147349834442139, 0.004988238215446472, 0.8912541270256042, 0.05228408798575401], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b6a52297dc1d2723ecb41bfd9cd99cbbca46c6c2ca842b22c75d0e8d59bb222d:action", "state_id": "7258b6e961e27fe42d7e01a234b894111a8d8ed27a97f2575215bbe708af5eca", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.7109375, -1.552734375, 3.490234375, 0.7421875], "student_probs": [0.05481433495879173, 0.005698938388377428, 0.8829323649406433, 0.05655432865023613], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "95ddfc1af3560b2afff553492ab7063da0a8124452fc2b9bbef37316dac0b657:action", "state_id": "44bc5603433957a4c90fd32a6c153e23658a35ba82742d184fcc6293cf15b8be", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.53515625, -1.34765625, 3.232421875, 1.677734375], "student_probs": [0.1304083615541458, 0.007299882359802723, 0.7118992209434509, 0.150392547249794], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7ce79295549ad1569753b5f5cd33154d54fce9be7cb9106dcdb78e17f5fd7c8d:action", "state_id": "fb27c2ab786f4d0569c343d139e43ca2a5513f3031830e77adfbfdda86e873cb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.177734375, -1.08984375, -1.84375, 1.732421875], "student_probs": [0.34557974338531494, 0.035789165645837784, 0.01683969609439373, 0.6017913818359375], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e3cf7f8cec5b9566b1b2bd23e532840eee9957cbdd9a24468761d223943e17e4:action", "state_id": "5179cfcd08e786113624e2dba467f4698eebae7ebc6cd65ac76ad4c02040f675", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.37890625, -2.44921875, -3.0390625, 0.90625], "student_probs": [0.6034444570541382, 0.013125134631991386, 0.007276756688952446, 0.3761536777019501], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0eff34356e348a42fe0207b75aed070d5512de0bc2d42f4443440cebbc763a75:action", "state_id": "2965681e534a4e435903c3468cb8daa626b04f6471d6b2ceaf61654b85fc02bf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.28125, -2.41015625, -3.1015625, 0.89453125], "student_probs": [0.5825098752975464, 0.014525996521115303, 0.007275653537362814, 0.395688533782959], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "30ecd91d23286dbdd2dc60f498864793a13f8b5231c814b474f8dff564c62e28:action", "state_id": "795f2ab26312c93f377d29d43f5f68d45a6316bb9b0181594e16393402f57a58", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.57421875, -2.43359375, -3.08203125, 0.91796875], "student_probs": [0.6466346979141235, 0.011751361191272736, 0.006144341081380844, 0.33546966314315796], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6fdcdc83c70721d5108021f48ca744ec4e5504151fd8445eb4e747195fe89b2e:action", "state_id": "fe5f11bf75dcd1334afc25f845364c25feb2f2a5c1eb3e880f6ecb002f27b057", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.36328125, -1.875, -2.421875, 1.6484375], "student_probs": [0.4180765450000763, 0.01640167273581028, 0.009492559358477592, 0.556029200553894], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1b45e21bf52b2609c6b223de92e0f209be100f9d7bfeafc0dca355cf8897c5ef:action", "state_id": "15003b6924304410b3fd5635dde3fdec4527428ccdc986505c9a49cd5b17c36a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.255859375, -1.4921875, -2.109375, 1.69140625], "student_probs": [0.37815549969673157, 0.024221934378147125, 0.01306675374507904, 0.5845558643341064], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6e44022f2ac9c1b9ac72c50e6ff067fe87105fe78c653c51fad7e2fe7246e52c:action", "state_id": "10b54bb75d08b5afd825874ad366c8e84215b57819f8b77aaac67141a26159b1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.228515625, -1.90625, -2.1171875, 1.60546875], "student_probs": [0.394231379032135, 0.01715298742055893, 0.013890912756323814, 0.5747246742248535], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "12018ee5046d51d74502f0c1f03d03afc1941e87e995ad7ea0948c77f94dc855:action", "state_id": "71471b73455794c8bcf7ff461d646a7a79a34234b98f1c7109ac71c0ab7f1ddc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.138671875, -1.9453125, -2.28515625, 1.501953125], "student_probs": [0.3973924517631531, 0.01819123700261116, 0.012950005009770393, 0.571466326713562], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8feb86fca23d1f3cadf9eb5ffb20cfce7815df63ec0a20e9b0a36dd06193765c:action", "state_id": "dc695904d06deefd9f6bc0de4ceff9d6c8f6d387cecb8b29585bd7a183235e82", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.47265625, -2.20703125, 3.26953125, 0.33984375], "student_probs": [0.05453290790319443, 0.003740116488188505, 0.8939763307571411, 0.04775060713291168], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c39161d618546f9cfebc0f5bbe3e81974aada1f224e75c0967b592a0cc77dc6c:action", "state_id": "cbda1a5a958a9fea906399aa9af8e5f38b3a836ae9843d901fa547534fb6e94f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.51953125, -2.03125, 3.23046875, 0.40234375], "student_probs": [0.058786142617464066, 0.004586535040289164, 0.8843417763710022, 0.05228547379374504], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "aa94cce2098dab2cdab35029d450e8b43308eabd2b54b8553fcb8c8e14686961:action", "state_id": "026bb6d0522ca10bb997ee75ffeba48de36552ba2b1b16dc7aabc19cf5011cc4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.7578125, -2.0234375, 3.310546875, 0.5546875], "student_probs": [0.06793335825204849, 0.004209219478070736, 0.872411847114563, 0.0554455891251564], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f3f216038dd81f1b560473d973e4d3017d4d254f6c5cb71ed858c91356052ccd:action", "state_id": "9ff65003f4ce581dcaba50a6e5db98c7e72a864b587b0429a34f4ec7d28afbce", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.40625, -2.1328125, 3.37109375, 0.40234375], "student_probs": [0.046583741903305054, 0.003677337896078825, 0.9033367037773132, 0.04640213027596474], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1f2fea8a2a7b333378a7c4a12a5538ff830f6b50d650a47f8171482238462be3:action", "state_id": "e34f922e88a9ec27a2fe7c908a5d86b049cf5c3da536ef337c730e139ca14842", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.546875, -2.03125, 3.470703125, 0.48046875], "student_probs": [0.04848703742027283, 0.003680952126160264, 0.9024602174758911, 0.045371778309345245], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5c190ed22a7bc682c2ef1e43448e3234fc22df957bc3a0975088495cfb4ec1ea:action", "state_id": "34e11b1bfe8f04c882fa96d804a57a8a7f9dedf8129d83e3bc1443bb2ad1f823", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.23828125, -2.125, 3.53125, 0.0859375], "student_probs": [0.03463146463036537, 0.003259198507294059, 0.9323714971542358, 0.029737794771790504], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "389f834f65d6acaf9f3b71be157842733dad7ad073db872fcea1326d0fce7bac:action", "state_id": "720f8f14e13c497433828c941a4ccf791d1dd7745f5d89d5c1f50bf20f3f8ac9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.36328125, -1.814453125, 3.5, 0.30078125], "student_probs": [0.03987099230289459, 0.004517300985753536, 0.9181564450263977, 0.03745533153414726], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2972b92e2c7467aab61124ec327e2b5b8c91c70f290dcdcef0bcea86d747ad39:action", "state_id": "a406aadb2a57a9e679441679eeb86be6477fc6a91201eaecf791afff76e9a05c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.666015625, -1.505859375, 3.49609375, 0.720703125], "student_probs": [0.05230957269668579, 0.005961394403129816, 0.8864790797233582, 0.05524992197751999], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8fb44a3f13feaee7d4826952015bd1bb52b32bf9354d931651109b4c975e8a41:action", "state_id": "3b1ba61584762ccfed521da84fd89445041c1aa19199cb25b4a2ae2f324a41bd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.65625, -1.521484375, 3.478515625, 0.728515625], "student_probs": [0.052622877061367035, 0.00596206309273839, 0.8848485946655273, 0.056566476821899414], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cf3839263c2ee8fe0e96ff6bd3fe2f7ac2386abdc5f0db30be7a3e89d0ed6912:action", "state_id": "172752a69277ce3c49ce8e513596b3a15f47b0baad1d9fbcb13e65f44f0ea250", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.564453125, -0.982421875, 3.08203125, 1.845703125], "student_probs": [0.14358994364738464, 0.011246833018958569, 0.6549374461174011, 0.1902257651090622], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ed95251bd4213d771940b623362f3207cf5aa2067aa99509ac67c424d5d3ab77:action", "state_id": "dacb9b4c712bc60d3a1163c088c400babb77b258ee7429adb5304915cb7a11b5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.11328125, -0.900390625, -1.81640625, 1.779296875], "student_probs": [0.31914442777633667, 0.04260501265525818, 0.017046693712472916, 0.6212038993835449], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "52b10317ac0fae39dc021672a42eac0c90fe09df34feed3d6b8c85c359de7f6b:action", "state_id": "e0a6f513772d7061466c724aefb7690a16f066b402a45ddb11f00514b2561cc8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.244140625, -1.8671875, -2.04296875, 1.654296875], "student_probs": [0.38625597953796387, 0.017204521223902702, 0.014431176707148552, 0.5821083188056946], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ac5c4a321f7c26588cd915bc05cd7bbbdbc1ca6d5aab65cf91beaad35013d723:action", "state_id": "1051b130d8ce84c11a45783cecb6ab90aae0e19de38cf359003f850654563813", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.162109375, -1.9140625, -2.2265625, 1.541015625], "student_probs": [0.3936116397380829, 0.01815948262810707, 0.013285761699080467, 0.5749430656433105], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "791eea1642c2e859e8cf6ddb69c999f1c3cc9467bf82106bc89e374dd14ee4d3:action", "state_id": "509ea2b34791308dcaa0ca41abd54188b5819b71876f38a38b71e3633c85b044", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.359375, -2.71875, -3.2734375, 0.6875], "student_probs": [0.6504418253898621, 0.011017962358891964, 0.006327083334326744, 0.3322131633758545], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8e152ebfe83f6f27e192c8b397491c339146fa3034579210894fab33f3cafade:action", "state_id": "e5592fcd2cee380cd57d09170257f175d37f47ffc0733144289a38a12e13db47", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.28515625, -2.66015625, -3.32421875, 0.8359375], "student_probs": [0.5997254252433777, 0.011601789854466915, 0.0059720897115767, 0.38270068168640137], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9020b26e5029c4f8d59f44241fc1ea1cca43d61614807c14bab2a57e57b6e477:action", "state_id": "cc60a2af528c55532130d8f55a09868f17665f8f7fc5825091865fccec2d6a67", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.58203125, -2.5625, -3.3671875, 0.66015625], "student_probs": [0.7038722634315491, 0.01115698367357254, 0.004989712033420801, 0.2799810469150543], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bc0bd44b943dda85cf928ecf0c16d1cc838e15edcc709b36107719f5b7fbcd28:action", "state_id": "382796f9fc1bc03faec07173ef7c68de9dcb331c422f421f941c69686b59eb84", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.4453125, -2.4609375, -3.14453125, 0.79296875], "student_probs": [0.6447062492370605, 0.012968778610229492, 0.006546633783727884, 0.33577826619148254], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1a0bebf005df4cad3d5a1233d266a7abc476967edbc6fc10a42d1d99d5891434:action", "state_id": "82ab7ce76f19637977880f1dac39204f0709772a3d4f90e009020f9d10065eff", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.4921875, -2.171875, -2.96875, 0.88671875], "student_probs": [0.6317126154899597, 0.01618964970111847, 0.007297246716916561, 0.3448004424571991], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ca9ef4d414547622069bb346f7dee2e860336cd0107748c4d42b7ae811aeda7c:action", "state_id": "fa9521ed23953ea015d26a6433ce9a4d547c0636840cf28a8a057b1523788c46", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3125, -2.2890625, -3.2578125, 0.3515625], "student_probs": [0.7041411399841309, 0.019209718331694603, 0.0072911870665848255, 0.26935797929763794], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b4bb20ac5b3f12a125ac0f7bc449f396c73e850a6d8f90c7caca3b84c29037a3:action", "state_id": "f198c14edc992d003a00182a7caffe5fde6fe89d17980b9b3183b8c6dcfd0338", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.140625, -1.43359375, -2.22265625, 1.65234375], "student_probs": [0.35984283685684204, 0.027424827218055725, 0.012458288110792637, 0.6002739667892456], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0eede9a8e2c8e1ee615a4f2d47f329c7e97abe12cfc8ba7d0c3a7536aa601122:action", "state_id": "5a455f5a07dc291abe2a0eeaca9d0644451c6711fe341dba4e1ea7f8a929c47c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.095703125, -1.35546875, -2.13671875, 1.5625], "student_probs": [0.36757519841194153, 0.03168223425745964, 0.014505182392895222, 0.586237370967865], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c349d93df77f0a3cd2a21cf95a8beaa7723fae0d44d7861a2886921c994c94bf:action", "state_id": "39c03680674e25ba088d04536412bf549ac764eba827e32c706d65c5705ec785", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.072265625, -1.41796875, -2.16796875, 1.609375], "student_probs": [0.3529703617095947, 0.02925790287554264, 0.013820454478263855, 0.6039512157440186], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e5b468952dfc5d065c3b122e41e0e18b7e7abcba8253255349616fd8a2ace213:action", "state_id": "f74c76c3e8e899df91277e0da8937158ee7917920d31dfab66bb7eb90572d034", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.36328125, -2.4140625, -2.97265625, 0.8671875], "student_probs": [0.6079479455947876, 0.013911913149058819, 0.007957793772220612, 0.3701822757720947], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3977e76b8603c7dd9e25b887376ec4a0642b7431a074f7d3d0aef1f907bc3abf:action", "state_id": "29b85a79388a0c79b565ed42e2ea8304f43c24bc8077cf4e10821b15ae0863bb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.265625, -2.39453125, -3.0546875, 0.80859375], "student_probs": [0.5980200171470642, 0.015386153012514114, 0.007951111532747746, 0.3786427080631256], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "64963ee641460fd90a2ec134e90b309eef9234f34a60056e8692999951aa9f54:action", "state_id": "960414ec02f2452ed772b6fdaf4f31fcd55a4d4530b55e8148891bc752803092", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.46875, -1.8671875, -2.33203125, 1.6953125], "student_probs": [0.4324856698513031, 0.015388364903628826, 0.009667482227087021, 0.5424585342407227], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4d2eb667d342f6ef990af0da0096d6673d1b5f236687bf0086e3eb5cef699ca8:action", "state_id": "fbb4232b76b7f4ba45e8214f7e784a249a7896793e60ea98c90a5696be44b3db", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3359375, -1.8515625, -2.34765625, 1.556640625], "student_probs": [0.4322715699672699, 0.01784197799861431, 0.010864061303436756, 0.5390223860740662], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4648df67d499035b0082bff37e3ee4fcda87b4ce5baa0801c316b39dd5f57181:action", "state_id": "90aa6ab4002d5c2077cba41172040985e71ab0a26416fd7d7a5765cfa6a8ce55", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.236328125, -1.9140625, -2.1015625, 1.60546875], "student_probs": [0.39606499671936035, 0.016965599730610847, 0.014064975082874298, 0.5729044675827026], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "37484f3d18100d58aefb42c0237faba2b06aca9033018eb1d1efd5292d67ca42:action", "state_id": "c87f7ae81361b2b4b225eb3e1569a198a87bcfc653634df4fad0b31ea257f940", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.138671875, -1.953125, -2.28515625, 1.501953125], "student_probs": [0.39744871854782104, 0.018052227795124054, 0.012951839715242386, 0.571547269821167], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9d2944288c4d0d30c4e1c5bd169f8bb7a51bca0cb61edfb6bea1be2281a06566:action", "state_id": "673f961dd83abaad6f20f9c9a8711f09ffd990032332b69ec522893c577e2030", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.390625, -1.9453125, 3.185546875, 1.587890625], "student_probs": [0.12088020145893097, 0.004301064182072878, 0.7275784611701965, 0.14724025130271912], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d6dec3214553bce4a94a384da012f45f4dc887f36dcb65023e9b077cf5c0cf5c:action", "state_id": "b8d6b542cb20d75e2c96943b05a526c211063e538a671f5f6ba601d6bfe3e2f1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.33984375, -1.74609375, 3.1953125, 1.27734375], "student_probs": [0.11933466047048569, 0.005452064331620932, 0.7631087303161621, 0.11210453510284424], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "16ba93859370e74e8966d507de12c4b640d5216fd721a1809b5ee4a20e3484f7:action", "state_id": "b626b5606216a779bc6249ae16f6ee0038215365f0557c7aa9dcc0ccae231fb9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.44140625, -1.625, 3.224609375, 1.369140625], "student_probs": [0.12617097795009613, 0.005878088530153036, 0.7505761384963989, 0.11737481504678726], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "9f891b966cad2380d9ef90fd86127344a63e5491079d6414f95dcee086a4a517:action", "state_id": "1a252fdc90f914ad2237bad7a8a51368b087d0b2bf34caef388ca4c736cbe472", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.072265625, -1.34375, -2.078125, 1.5], "student_probs": [0.37510916590690613, 0.033488478511571884, 0.016067948192358017, 0.5753344297409058], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1858561f044f31538161ee511741d343c54798da9941650fa7d8ccb10cf9cade:action", "state_id": "c97dbefc325f678a06d2072afce427ee4ae37c3c15100d0a73ac8710648ff72b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.080078125, -2.51171875, 3.546875, -0.208984375], "student_probs": [0.025274841114878654, 0.0022214693017303944, 0.9502856135368347, 0.0222180113196373], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "17720fb115bbb63ac5c724de4561929560fb13c1fea74e18ba36c78f808bad0a:action", "state_id": "af3a5dc7646a621f5834e2f14584d74a13777e9a397bceb1487f95d0ec31fa47", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.05078125, -2.25, 3.4765625, -0.056640625], "student_probs": [0.030539032071828842, 0.003059416776522994, 0.938973069190979, 0.027428528293967247], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d33740d7acf9dc9290fa298c2c76f89fced53ffa94b2c21debd9fb86fe442280:action", "state_id": "6404f92858fcecc4cb1ce292652c3b7996da12ce8e694c966a54c5170b2c3f74", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.20703125, -2.076171875, 3.53125, 0.08203125], "student_probs": [0.033600181341171265, 0.003425776958465576, 0.9333219528198242, 0.029652057215571404], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d4ea04df5a510c37e0c75b09bee0c9eb831e68a95531cf41bd21d6d38092f1d1:action", "state_id": "99c8200849b7b1d6f9b4e8ad1fbdb27dc0b0305b9430d61c137ffb6c867d4b4b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.6015625, -1.73828125, 3.4921875, 0.609375], "student_probs": [0.0497296117246151, 0.004791084211319685, 0.8953596949577332, 0.05011964589357376], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bc6043448f338b2fc86c5c34c691e928821496683c58c55507c59dc4c68ecb7b:action", "state_id": "85ad1463f55933f236cd663837e4447e66b3014a1ec43172a30124210e990b2f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.669921875, -1.56640625, 3.50390625, 0.693359375], "student_probs": [0.052236564457416534, 0.005581483710557222, 0.8887065649032593, 0.05347532033920288], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6c243456726ed9396ebd574d18f9d1253ca0d2ff2af20bf32e603377c1a1898e:action", "state_id": "6717e8a4d2c20eb0afbdd6dd419ad14f0470b8a0be79744dc0d79f74e94ec0fd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.568359375, -1.3203125, 3.21875, 1.6953125], "student_probs": [0.13513463735580444, 0.0075202519074082375, 0.7039182186126709, 0.15342697501182556], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "50b2c59d7575bb226d4b4513ce0ecfe2acf48eeceaff0987758f0c3a287ae2f6:action", "state_id": "1b744117a44370940e44463e536d5b088b3e2ba719f8256436572574facbb400", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.185546875, -1.0703125, -1.8515625, 1.732421875], "student_probs": [0.34714964032173157, 0.036375537514686584, 0.016653936356306076, 0.5998207926750183], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "fe9fd3b4c25359b8ce7a83d7a203b05651bec860fef073dc10e62487c71f8dea:action", "state_id": "bd3675c7c8af72b3a07fa297ee94c84c0910d96780961c513a545a95496174df", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.35546875, -2.73828125, -3.02734375, 0.9296875], "student_probs": [0.5943797826766968, 0.009912220761179924, 0.007423910778015852, 0.3882840573787689], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "be40770881a79b87481898b04e081f836f90fd320fa5c3184186f7e2de7c81ef:action", "state_id": "3cdaaf680d0d985c70b420c58cb03e853977818258f93f433e552f104041f351", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.2890625, -2.734375, -3.2734375, 0.6328125], "student_probs": [0.6463620066642761, 0.011564292944967747, 0.0067453933879733086, 0.33532819151878357], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a8450cbd0218f5a45b7beb87e4b338475a504fba9ba4b5b18d1b1c963cd17d69:action", "state_id": "a862887c489015d9db40daa85942b7b224af505d974bc35b6dc2739582c580fd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.58984375, -2.53515625, -3.19140625, 0.98046875], "student_probs": [0.6376577019691467, 0.010306776501238346, 0.005347085185348988, 0.34668847918510437], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "28f8f64facbdd136f62b9b52478ab81e29ad55d6579a0c8f916cbda36c3a4cfe:action", "state_id": "bb137f053529833215a5f894d7a877b651dcb2c4f880531d2aac7bb515208617", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.48828125, -2.5078125, -3.3515625, 0.56640625], "student_probs": [0.7022135257720947, 0.012911828234791756, 0.005553307943046093, 0.27932125329971313], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f03f0455e5ba8c72ec99adc7de754f1976c3b83e469687353e26f0c582864e82:action", "state_id": "6571f9b46d6badc5686ca6896526711ccf6b979e792598d1fa2b511754582481", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.42578125, -2.1796875, -3.05078125, 0.6640625], "student_probs": [0.6642706990242004, 0.018051359802484512, 0.0075543783605098724, 0.31012362241744995], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f41e808de4092c88772fb056742aae2b40cdb0efdcd0590f1853e1ac2919f5a3:action", "state_id": "9b666bb34bc177444d594a782f690963edfed459df0dbe6b8b1ceac05e3944eb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3828125, -2.3046875, -3.20703125, 0.484375], "student_probs": [0.6932916045188904, 0.01735621504485607, 0.007039991207420826, 0.2823120951652527], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b3470efac8407751858de05179669816caf6df08e0c048a36bcd40cf3c108f21:action", "state_id": "149bac9e30e5d3844a65385abc0274c74672788bb0d7efc86fb23f50701351fd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.04296875, -1.58203125, -2.3515625, 1.486328125], "student_probs": [0.37538549304008484, 0.02719283476471901, 0.01259654201567173, 0.5848251581192017], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "12d02ab395228c2bd2f2f3640249b0eb1f01567fc23248f28bb9d61348ae0411:action", "state_id": "708740e2e22c167706838aa893743d4015a04470ece435051ad19e20b7df53f4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.056640625, -1.40625, -2.16796875, 1.529296875], "student_probs": [0.36640647053718567, 0.031213561072945595, 0.014572466723620892, 0.5878075361251831], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "715e04f128107ee74e75b7eb535d98e591504e04c3ecabe02a3edc1dc50d88ab:action", "state_id": "aac40ee5b5c5a5ec4d4130b1cbc93ce4c75f3a2f3ca148cd0fd052a3a15a679c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.30859375, -2.2578125, 3.5859375, 0.26953125], "student_probs": [0.03503390774130821, 0.002690992783755064, 0.9285833239555359, 0.033691778779029846], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "93f00e1ff3ae9dfd43c98fe84b5eb066ec5a23eeb87c222ffc4c530e682cb651:action", "state_id": "94046aff6e2df59d0fe5204e0f0c47276e4dd5744bc1d8b1b9827b1a453d54eb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.33984375, -2.08984375, 3.46875, 0.29296875], "student_probs": [0.04017476737499237, 0.003537964541465044, 0.9179521799087524, 0.038335029035806656], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "11fe5d30bf6dfdd9a74962e8cf08eecb079a469b19c6ad7d31f3a5c7c9369ffd:action", "state_id": "81ca45709d96e3aeb8ec304128106c9da9b5543b07950742896a2ae677be3eef", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.494140625, -2.0078125, 3.50390625, 0.470703125], "student_probs": [0.04475994035601616, 0.003666950622573495, 0.9078500270843506, 0.043723080307245255], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ceaecc82ce64853447ac4b39fe509d407f1329d6411ee41446e0fa8f0aa326f1:action", "state_id": "df7d6004954f79647c5230bd93688b40720b103cea0361a765d93af8a6293422", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.388671875, -1.95703125, 3.5078125, 0.400390625], "student_probs": [0.04042936488986015, 0.0038723177276551723, 0.9147922992706299, 0.04090593382716179], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7a6b2fd61f89ff80a7ed08158018166cd94d944cdc5f87dbf9c967d3de57fd74:action", "state_id": "a090ebea96465475ebdf66c612f09187a1301424015f5abda9bc1e288dbb74f3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.67578125, -1.9296875, 3.478515625, 0.70703125], "student_probs": [0.053777072578668594, 0.0039724321104586124, 0.8867664337158203, 0.055484142154455185], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "da3665c431cce761bcef688ea2bae43b50a1e1b7c4dcca81758465ee143f8130:action", "state_id": "f021af3be54c486870fa048875611a2b53a96a535ca8c65355b5fc1cb72a6276", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.576171875, -1.80859375, 3.5, 0.501953125], "student_probs": [0.048466090112924576, 0.004464239347726107, 0.9020704030990601, 0.04499924182891846], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2d097dacae19176f6a9644d255083ef3df2862f3060fdc7fd1e4e1ab7786a632:action", "state_id": "49dd7f2cdee8577ef20b5f0f0f62e9943bfa0f9fc25dd0e4ae18e7a87f7f7e66", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.607421875, -1.1015625, 3.134765625, 1.83203125], "student_probs": [0.14441758394241333, 0.009618849493563175, 0.6651767492294312, 0.18078680336475372], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "aee804cc68f14f60af3215bf0b2398d755cae324076a48049fd09f7f7b6c8b47:action", "state_id": "bebe69c6b6da303e3469d3e4015f338ec643a9457e7357735558f81bc11780d6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.20703125, -1.033203125, -1.87109375, 1.74609375], "student_probs": [0.3488115072250366, 0.037125248461961746, 0.01606120355427265, 0.598002016544342], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "65c63abdd0286153518385e583684ffc19aa9ef0c546240c395b091bc5e5f6a4:action", "state_id": "f55d621950978ba68a57b9cd2ed465c10766be82f89e3840b0b973edfb23dcbf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.1796875, -1.04296875, -2.0234375, 1.66015625], "student_probs": [0.3615606427192688, 0.03916460648179054, 0.01469202246516943, 0.5845828056335449], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dae2c8de33d42ef4724165fe5c7a682f669043063d0b2b49faf487ffe0eca663:action", "state_id": "8aec59b33beec67d4e4ff4efcc2a90c04d968e3397881e7e773248d69e83974f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.431640625, -1.9296875, 3.201171875, 1.63671875], "student_probs": [0.12299499660730362, 0.004266592673957348, 0.7217471599578857, 0.15099123120307922], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "04ddd6d7d802fe57dcee74f3545d51afa19632d67c7a1bedafe98bcbc08d57b2:action", "state_id": "b0294aaf259defce54f57eae1c710ea663317fc556fa0bde601c46b94478ecb5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.48828125, -1.85546875, 3.064453125, 1.685546875], "student_probs": [0.1410481482744217, 0.004979608580470085, 0.6821660995483398, 0.17180615663528442], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "243161fe7403458b2b041ac744630ed4f1064e40209e35cea76352fa51619862:action", "state_id": "5bc88338175e56b57e64942004ad66dc1d4ac811cc9bd6be797dca79ba552262", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.40234375, -1.78515625, 3.193359375, 1.34765625], "student_probs": [0.12525686621665955, 0.005169968120753765, 0.7509823441505432, 0.1185908168554306], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7d48999e4e4c92faefe70e3eabe69144aa2dba5f160cfed658dfc012f9c98eed:action", "state_id": "ab59ba57f06bd106e310bfffec64cef71a02fd091bb5fd33ca5bc8822d62e988", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.400390625, -1.6640625, 3.220703125, 1.3359375], "student_probs": [0.12257836759090424, 0.0057218801230192184, 0.7567727565765381, 0.1149270310997963], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "36b3ce18e988a8e9e58042029a14f1abda4aba90ece3d50eb9e53ef3bf1159f8:action", "state_id": "92a1d4933a439674f064b5b8bbeb6b55579d43141dac3c78a6573534ef7c9970", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.140625, -1.18359375, -1.984375, 1.724609375], "student_probs": [0.3407194912433624, 0.033342763781547546, 0.01497016940265894, 0.6109675765037537], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7158a9480bbcf22d37efdd19f54223fdfd62e8079a116d205f58094339c82137:action", "state_id": "1cc063f8bbde6b87020e2e802a34cc91c1207cdf6518e80ea6515a272241317c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.228515625, -1.8671875, -2.05078125, 1.654296875], "student_probs": [0.3826017677783966, 0.01731012389063835, 0.014406763017177582, 0.5856813788414001], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5a5d29876c853dee54a9cb0edb1a1a55d5b8b300cffa9e5b9f720f8aca1ea11f:action", "state_id": "45358902bf3e5d38b6cea720f15fa3acc6d6ec7b22e30962ff38eea71959dac6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.146484375, -1.9140625, -2.2265625, 1.556640625], "student_probs": [0.38636884093284607, 0.018106039613485336, 0.013246661983430386, 0.5822784304618835], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2d90103a88855395f24bb142667d0e7a151e898c272417e21b75c86910241175:action", "state_id": "029646c8606ed24bbbaa69126570efe2b0c6b633763b821f00b32fffcd15e406", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.455078125, -1.921875, 3.1875, 1.66796875], "student_probs": [0.1261713057756424, 0.004308921284973621, 0.7134144306182861, 0.15610536932945251], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "de40e518029ed44a83871c18e1d7e94921511f3f8a38403e02c608b59b38c139:action", "state_id": "1d91d137df7b881d681a463efa2c37cfcdf942f475f94d9deccb6c8307006815", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.380859375, -1.73828125, 3.1796875, 1.33203125], "student_probs": [0.12439200282096863, 0.005497520789504051, 0.7516463994979858, 0.11846407502889633], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "016769ec5a71c7fc56c09977d2ba6e8d0be0539ba88019056e18b30e9b9bcfca:action", "state_id": "7396f00f37526d5939863d08fa13cc1d18bfc299a3e3605e47a46f7d9510acd9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.091796875, -1.66015625, -1.9453125, 1.431640625], "student_probs": [0.3973765969276428, 0.025353869423270226, 0.019063493236899376, 0.5582060813903809], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "18014cae2941f5ab497120f04a1abbc08ff93b801cd65fd6b973e306563be4d2:action", "state_id": "33f8192ffe4ddb82e1e9596d701f322ae3008f55b659bebb9483bbb1df9743f8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.447265625, -1.9296875, 3.201171875, 1.63671875], "student_probs": [0.12469036132097244, 0.004258344415575266, 0.7203518748283386, 0.15069933235645294], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c5def94fc5e4a882780f70cedb65607fef4cfd96bb433dba134f56c265023616:action", "state_id": "d8c50fe4cf579db9ef3ece4ade8576a6a9c270fa96211072b46f8fa2c7faf81d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.365234375, -1.74609375, 3.1953125, 1.30859375], "student_probs": [0.12159755825996399, 0.005416169296950102, 0.7580846548080444, 0.11490161716938019], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "514fd00f3823e71413c447a7173c4ec372fcf2a2df3d0535a24a089c61e61b67:action", "state_id": "d3a19de524e406050c91897bfbb570677ad4db67fe4d2a3867fd06d20e1b7d15", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.44921875, -1.625, 3.224609375, 1.369140625], "student_probs": [0.12703484296798706, 0.005872277542948723, 0.7498341202735901, 0.11725877970457077], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c2de9757ebf7836774c8f0d2a4a219b02a5c76d714d6a2f5ba0e883f8edc78d1:action", "state_id": "75ba6a10b1b0ca63a91eb8bb04a847843a61018ccef1d8cebbd297d1aaad9471", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.123046875, -1.36328125, -2.03515625, 1.59765625], "student_probs": [0.3658844828605652, 0.030447062104940414, 0.015550837852060795, 0.5881176590919495], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "be933dae950d2b34958b9f2050c91b04a1bea520e720a8777a2e768b80e53a6f:action", "state_id": "f7fbda94ff43a25386a44e0e23670bf44761a54a22474edc040bdad906fdc96e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.16015625, -0.982421875, -1.9296875, 1.697265625], "student_probs": [0.34795746207237244, 0.04083346948027611, 0.015835218131542206, 0.5953738689422607], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6e110394b5e36f9f21c73a98b143de5ee0d0c80dc51ebca06c4edbeed1622194:action", "state_id": "a94848a4543021d3f7160923730c8e7f9b98c81f5da43ea5d8be5dcaef5219e3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.26171875, -2.2734375, 3.5859375, 0.23828125], "student_probs": [0.033519502729177475, 0.0026563983410596848, 0.9310809969902039, 0.03274302929639816], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d53707bd815f58e98c120ed3dd8606066ef209bf8c7ca06e94aabcba7d308c09:action", "state_id": "8a08710aa31507ba00163c90e6c1a1be1f276665e1eea3d625e9579200c7dbfe", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3359375, -2.07421875, 3.453125, 0.30859375], "student_probs": [0.040575187653303146, 0.0036437029484659433, 0.916300356388092, 0.03948074206709862], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3f48c87a8c7e8e3a35a1b73224df21fc44122086ee348353f0dc333bae7f6396:action", "state_id": "1775f59ab54812d0f9d7d5770dd52fff0468abf357caba7f7c83ed051f32674f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.345703125, -1.935546875, 3.509765625, 0.341796875], "student_probs": [0.03881264105439186, 0.003964961040765047, 0.9185611605644226, 0.038661323487758636], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e53073404e01b575a38f86998be84b79fb23fe6a109b15dea7b3e6f9201e414a:action", "state_id": "24f26ed33eda7d2e87484fdec045f3fdb149e322db341fb525f0613e42238081", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.71484375, -1.96484375, 3.4453125, 0.67578125], "student_probs": [0.057569365948438644, 0.003948370926082134, 0.8831183910369873, 0.055363912135362625], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a68bf8c8933f45cfc862ca89126a12d4dd741494180bcdd6488b231f073f58de:action", "state_id": "ecf7c9b0024668d70fc50bbb022ae58f4ec840768f5ea2df49392f26545df289", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.607421875, -1.802734375, 3.501953125, 0.638671875], "student_probs": [0.04951335862278938, 0.004446362145245075, 0.8949552774429321, 0.051085080951452255], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "07284e080073047a7383a56951321ecce93f31cd55c6c9f7fb91e2bd5a7b40e9:action", "state_id": "5a2c4ca95946fde0c47acd00705546f84f32b16a9c22220b321ee28166259b7a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.5546875, -1.27734375, 3.22265625, 1.69921875], "student_probs": [0.13305340707302094, 0.007835927419364452, 0.7053677439689636, 0.15374290943145752], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f3f66942343b9ce80c46ee15b45a144118b4799afce1f3829a39063942367a82:action", "state_id": "a42e02ee8fbf87aa4a4187e5e0d2c82fb17f6653a5a2d4e5c16907faab811472", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.201171875, -1.0703125, -1.875, 1.73046875], "student_probs": [0.35124292969703674, 0.036233846098184586, 0.016204778105020523, 0.5963184833526611], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "51ef8571b5eba031a349cf6fe3826faea75acb3d956d67c7fbdabe08229a3a0c:action", "state_id": "a81e53b39f4970f8ef28840c62548d74b79b1be409bf49ae0e7a3e28cca4b971", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.212890625, -1.009765625, -1.9375, 1.677734375], "student_probs": [0.36457476019859314, 0.03949110209941864, 0.01561670284718275, 0.580317497253418], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "df689138e02dc64eeece35446ba3f4168c6edcf39162632855342cee25fc93f3:action", "state_id": "ae4b1b33c822f7a9d6834e9cc42899c659bb0bddde20af77aaf9434716fe1031", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.38671875, -2.703125, -3.01171875, 1.09375], "student_probs": [0.5633536577224731, 0.009431581944227219, 0.006927299778908491, 0.42028743028640747], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5bff55c5b47b6fa8f89e0423b43286296a22cb1ad47536bde5869b6fb1f6d45c:action", "state_id": "fa98eba08bb8235a371bb45d0c4d7f9822e8608e598f075a26c7c9dbfaee6c0f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.32421875, -2.5390625, -2.97265625, 1.12890625], "student_probs": [0.5384485721588135, 0.01130687352269888, 0.007328838109970093, 0.44291573762893677], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7b41f43dfca2091fbb8e096c937f2c90c02c132d1759ac3c6677277a39c59528:action", "state_id": "256224bbc34c9577de743bd275f0491c29416cf3418d366f456a4baa9b96c874", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.62109375, -2.46484375, -3.03125, 1.24609375], "student_probs": [0.5835545063018799, 0.009808019734919071, 0.005566653795540333, 0.401070773601532], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d7721ca19dfbbc4f0d490244d5001431c2643c3af675033e71a87cb6a28eb6df:action", "state_id": "14474ae5011d01dbfb53bf051ad0624a0c8ff9f5762d538900220fc476470364", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.4765625, -2.5390625, -3.2734375, 0.6796875], "student_probs": [0.6768561601638794, 0.012204854749143124, 0.005855953320860863, 0.3050829768180847], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4e7742a3910e6fdda47ebdc455f23b607c0122d89544b139b7a47cfa6760d7a4:action", "state_id": "3442165877f1c641095cd33d77408de06823b3ea16e0024137e2dd929a499363", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.4296875, -2.20703125, -2.97265625, 0.7578125], "student_probs": [0.6454373002052307, 0.016999932006001472, 0.007905702106654644, 0.32965710759162903], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2bef19dd5f51d7c4726215e51cc347f59c8acd3f51e1922d8fc8655fa263e8ed:action", "state_id": "1d03c2cff52cebecb87ebd799d020693b9b846f8cfad7cfb2254052dbcd600cc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3828125, -2.28515625, -3.2890625, 0.40625], "student_probs": [0.7084729671478271, 0.018086088821291924, 0.006627561524510384, 0.26681336760520935], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "42cbc2477a2f44f00134ff0e9e4e7f7a8f5b898b62c45718b58300445489d289:action", "state_id": "b3c57a28ba87a6beb61f4dde786e50aba35e22012b1bb12f35f345e349c3b9c2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.19140625, -2.10546875, -3.0234375, 0.5625], "student_probs": [0.630935549736023, 0.023343736305832863, 0.009321839548647404, 0.3363988399505615], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "26cf24c220dfcd5df17df84a6cd764b346d8e040b5abd9ebc525d92073838412:action", "state_id": "0990f7d0853a4d29a89053cbdeb0de7e2865ad07fae868ac4522a2fd2feda04c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.044921875, -1.46484375, -2.2734375, 1.494140625], "student_probs": [0.3724968135356903, 0.03027925454080105, 0.01348892692476511, 0.5837350487709045], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0360295a9b58d1dc6e357a07bf7965d8ff66f60f33a3871fc3221f6ed92abbea:action", "state_id": "ec23ec1187e23a605965dbe7d884435182636afe7100b93aefd18f365f4335d7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.10546875, -1.53125, -2.08984375, 1.5078125], "student_probs": [0.38345399498939514, 0.027453698217868805, 0.015703869983553886, 0.5733884572982788], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8dee78bc0e73fd5f59c91f0921455ba2f21a8e4e71ebddb6c691605205236be7:action", "state_id": "b4a5f8573e4bf44b647d4f5b8be959705a3da94abd21478ae5c22f1d70f66445", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.205078125, -1.890625, -2.07421875, 1.654296875], "student_probs": [0.3773605525493622, 0.017072996124625206, 0.01420940738171339, 0.5913569927215576], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "05b3b1c186ef6ebb4e93f8dde1ebdd9963b19ebaaad50aeed419d7680d822c4d:action", "state_id": "0cf2fe094f101376df5003e33d55506dc86f66c87d1f581b3d7b8bf67ed37062", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.123046875, -1.953125, -2.27734375, 1.525390625], "student_probs": [0.38837650418281555, 0.01791795715689659, 0.012956332415342331, 0.5807492136955261], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1afae698e79e2f658ebd708dad8c1a18c8a8e96a9a5f4d9e11509b3de19dba41:action", "state_id": "a09f7a61a9621d635014115a9089a190b764e0a8642ac19417ece53c0bcc98ef", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.236328125, -1.8671875, -2.03515625, 1.654296875], "student_probs": [0.3843619227409363, 0.017254430800676346, 0.0145865548402071, 0.583797037601471], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f7f1682b9c1ebafbb3b5c1c7954fdabed3454f868615f90b1d3cee10a38d2b64:action", "state_id": "d88f78715a52aea569bff4697c06eee6da39b312a9786fda22c5f13e4e6716dc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.154296875, -1.9140625, -2.2265625, 1.541015625], "student_probs": [0.3917485475540161, 0.018215280026197433, 0.013326583430171013, 0.5767096281051636], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "264c7c325241cc229b0eb951b149cb5b4b19ebb2bc221d1da7fba06af536e7a0:action", "state_id": "b400a1506e4e1bac966491d41c43d543b9f3b83836599fc6094fa04f3f0a1559", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.416015625, -1.9296875, 3.201171875, 1.587890625], "student_probs": [0.12220044434070587, 0.004305785521864891, 0.728377103805542, 0.1451166272163391], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6e2fb1c12abb9cde207e18c146f79bf2ecede95234154d4c0fd867dfff47cf5e:action", "state_id": "0fa5d00ce5dca5cdd2f800bdfb0b969b6ea03ae4ff92cbf20c41f5bb83fe4dbd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.34765625, -1.74609375, 3.1953125, 1.27734375], "student_probs": [0.12015814334154129, 0.005446965806186199, 0.7623951435089111, 0.11199970543384552], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bc7f45ebbc431643f3cbe05cd607d7511c5801a692bcc977f4050d390f649feb:action", "state_id": "9dae685a0340daafff6a9b6e637849b6193e75b062bba75011520aa324ac3af2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.44921875, -1.625, 3.224609375, 1.37890625], "student_probs": [0.12688882648944855, 0.005865527782589197, 0.7489722371101379, 0.11827339231967926], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c28eb657190561d9ab465d6053514e78fba061f57bf167ed43414550f38209c8:action", "state_id": "b504a082ae095829b5fca4489176454f4ab3fb3fe5884ea75d7ead1ba504b647", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.146484375, -1.33984375, -2.02734375, 1.595703125], "student_probs": [0.37145093083381653, 0.030910275876522064, 0.01554266270250082, 0.5820960998535156], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b1a696fd80002f4a611185748bb2f57517694baee882118a36fd93f33eeffd28:action", "state_id": "e24b0b7d21b948a1745803446a8f277006bfd64573a45293519ba12657d395b8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.16796875, -0.982421875, -1.9140625, 1.697265625], "student_probs": [0.34964513778686523, 0.04071221128106117, 0.016036823391914368, 0.5936058759689331], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "91983bf5092b2db074758a53c4ca932c34208bcb576f8826fe16c68ed1f85f82:action", "state_id": "3696a26c85a47f3f1f2b136954dc4ecde7c664741d997c7f9fcc286e02897cd8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.33203125, -2.265625, 3.5859375, 0.29296875], "student_probs": [0.035807106643915176, 0.002665762323886156, 0.9270917773246765, 0.03443535789847374], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a818e4caa8d7fbb9fb9f33d7064c3844eb9e597980a623cf44743b67afa3ca3b:action", "state_id": "e75cf25126dee9c70b5e1bdf63d1463adaec62eaaa35acfbbdb2a82b3f136490", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.36328125, -2.05859375, 3.470703125, 0.328125], "student_probs": [0.040954191237688065, 0.003634891239926219, 0.9158715009689331, 0.039539411664009094], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4bb068aeef358ef0ea64ead4c8d154a98ba76b1b7ac57ab9a27e7c40f530b853:action", "state_id": "bda1ca6ca5f447f2b475862e585411e80aa0663f21e7bd1768c9bdf56a4a1b70", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.345703125, -1.93359375, 3.509765625, 0.326171875], "student_probs": [0.038835618644952774, 0.003975064493715763, 0.9191049337387085, 0.038084469735622406], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "03f6e374ac95c2b0ef0a4d1b8c2e28ea7a76471b8842463e597ee3f6df06782b:action", "state_id": "ff0011c3f1eeffe7f4aeebc14f971bb0fd0e92e354eb3a97167b0b8829d9b00f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.697265625, -1.693359375, 3.4609375, 0.712890625], "student_probs": [0.05566290766000748, 0.005097188055515289, 0.8827004432678223, 0.056539472192525864], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5540807dc215154a16f5ee228d71daedf0cf6edd6f735ae476ad3c7308d9fbb9:action", "state_id": "ca4c0e804539ef81cd215d80fb9089c423d18c6506a3bc90f9e8923a1594d9e9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.693359375, -1.56640625, 3.50390625, 0.7265625], "student_probs": [0.053313031792640686, 0.005564544815570116, 0.8860095143318176, 0.05511290580034256], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7f3fa4d9cec3557f6dae214ee433646340396f86928be553a759826119b774e3:action", "state_id": "8d19cc14725fe64c7c8176c8776d354f00e103c8fa0b3a9822420457dc599a07", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.107421875, -1.22265625, -1.98046875, 1.62890625], "student_probs": [0.353680282831192, 0.034408897161483765, 0.016127126291394234, 0.5957837104797363], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "54a87031e18aca7c6e7161b4cbf37782bbada403e976ca95a6e880b62e06c872:action", "state_id": "2b66177b165c34df1bfbbad47dcf515b38412179ba8549a3575c9e39886a3403", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.14453125, -1.06640625, -1.9140625, 1.681640625], "student_probs": [0.34872305393218994, 0.03821929916739464, 0.016373829916119576, 0.5966838598251343], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e03a34205b19953987a3764059547f7647fd1ebce6d4d37639760bfd2141a5b9:action", "state_id": "c0f24b741cc5b8d0fddc2718b28f0ca885140ed668197d5320b7a79f0778c3cd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.359375, -1.9453125, 3.201171875, 1.541015625], "student_probs": [0.11704453825950623, 0.00429678475484252, 0.7383008599281311, 0.1403578519821167], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "347c22042a958f3055e14b7287611f00d19d03414e160df6a144c7f5e7c05614:action", "state_id": "175f48011019b072267daa2b6f553397902fbb30f7c1e2c90ae04266623fc6ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.9765625, -1.79296875, -2.02734375, 1.228515625], "student_probs": [0.4168716073036194, 0.026134256273508072, 0.020673898980021477, 0.5363202095031738], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "87970bf6c574a01c02846cf087d1712cd5e9b4dae3911c7f3aa244fdf91b5b75:action", "state_id": "a16c9d4a67b5d7ded80382008aaa83344f8a43eed9e853d8c52bda4d73695be7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.115234375, -1.60546875, -1.99609375, 1.392578125], "student_probs": [0.41152429580688477, 0.027090005576610565, 0.01833001710474491, 0.5430556535720825], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2d50a6c935319136c634e645319afb5f8fe46f08314c83957a07f38bf86a8c85:action", "state_id": "61b484a07f8a3f3a50426ff7b8559ad43b43b86f1c807bc8c7a90c6032788f70", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.220703125, -1.921875, -2.08984375, 1.607421875], "student_probs": [0.39187872409820557, 0.016917934641242027, 0.014302087016403675, 0.5769012570381165], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a66dc5e63762d8bee6eb2abaa30e2444a9225fdaf1d9ad2f3776aec9d8a99e72:action", "state_id": "ea1fe5bd53398475aed065a5e3d1dad69dd5420d5d48ac480f0fd0e4380778cd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.130859375, -1.97265625, -2.28125, 1.486328125], "student_probs": [0.39924749732017517, 0.01792266219854355, 0.013163819909095764, 0.5696660280227661], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "739b9fa2dd4a989dd0fdd8571563801bbfb195bf697d4ca9a3eac2fadb8869ba:action", "state_id": "d469ce6a381fb68a8e4d1cf4ec2da8d2ad0b4a255248a1cfd2c5b19cf7f048a1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.02734375, -2.50390625, 3.578125, -0.177734375], "student_probs": [0.025810889899730682, 0.002168930135667324, 0.9498131275177002, 0.02220696397125721], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2ca1c5624f5ac84683e93d6e7e4c374cd06c650ea017c6131a14c94ea10b0c92:action", "state_id": "c2d82098cdf2ac3c78a1b69a511154b588d71200fb1053b77364e0ee7244f652", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.0703125, -2.28125, 3.4765625, -0.056640625], "student_probs": [0.031125539913773537, 0.002963782288134098, 0.9384961724281311, 0.02741459757089615], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "799071ff53af83f0bb4acaaf9d5472707cda757f99dc5cde72145af1cf821a59:action", "state_id": "6b69f6cf44decd530f588b0f2c76ee7f9868c0183084f5fc0a8462f23e378012", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.1171875, -2.1171875, 3.498046875, 0.046875], "student_probs": [0.031811490654945374, 0.0034057069569826126, 0.9351312518119812, 0.029651569202542305], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a9c65bfa61ee553eccf5a25d971e5ba42033153f4b101f2e06dccdef811cffa2:action", "state_id": "3a92314ba03b6df70f89e83419dcd9ea11a3224f3094ff253a9abc29c4340809", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.40234375, -2.21875, 3.447265625, 0.28125], "student_probs": [0.04354061558842659, 0.0031664161942899227, 0.9147182106971741, 0.03857484832406044], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "eb38e12eb1123be48e942fbae207affb094f3f119b229f0f970f644c3d730e58:action", "state_id": "7bfa87d525c8b5a92736b1376ec0aae1805c4b352aeb26c4480a338a19e602a9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.21875, -2.21875, 3.462890625, 0.125], "student_probs": [0.03618265315890312, 0.0031616047490388155, 0.9277111291885376, 0.032944679260253906], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cb46398a52cc766f08bb6d878feaffe6f5bf71a391c196fb239bee3529322227:action", "state_id": "b6c529fc95cfd529e4cedfaf1528c3bda1a80a7246bf0d96078831d3b929ee12", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.359375, -1.900390625, 3.54296875, 0.166015625], "student_probs": [0.03837021067738533, 0.004004888702183962, 0.9260008335113525, 0.03162418305873871], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "68263a9fd33f843350224abce8deddbf6ae52d3d061446b94c8c3d68326205ee:action", "state_id": "c521d3226f6d2a509fb265c762d3b961974c84a68ac673fb8d35cc49fa30db2b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.5546875, -1.755859375, 3.458984375, 0.5703125], "student_probs": [0.04909816011786461, 0.004870879929512739, 0.8961595296859741, 0.04987134411931038], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dca809c10fff6781dbb0d3112ffde151711b33bd93c5cf7ab887e543e20c78fb:action", "state_id": "3a8832778ba711c92bbd3efcb945a5d264d40cd53894ed87432b2710c5385317", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.703125, -1.521484375, 3.462890625, 0.728515625], "student_probs": [0.055772557854652405, 0.006029550451785326, 0.8809911012649536, 0.057206787168979645], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e3380b7eb4f385c5318911f54796c1e0fccc05a3558a453f81c2fb8175c319a8:action", "state_id": "86dbd311e0868b616825891c2d6e6cbfd19bd10f511818979a81fb854ef17954", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.60546875, -1.14453125, 3.2109375, 1.845703125], "student_probs": [0.13669289648532867, 0.008738485164940357, 0.6807571053504944, 0.17381146550178528], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "14ac2fc659e4ffafcf2db5e2af7b6e8e0ded4820638cbde87787a98d03363b9f:action", "state_id": "91951e5c0bfa0bc006171c8be3466b27818ecc59cfd014354d79c82570ae94c4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.1640625, -0.89453125, -1.79296875, 1.767578125], "student_probs": [0.3324311077594757, 0.04242929071187973, 0.017277436330914497, 0.6078622341156006], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b1f24be5f92a6c6b0b3bd650c296e82eebaa5b40949ce3808f5b2468efb1532c:action", "state_id": "6b8dcc5c78368dc42284e40eccf88a01773c92c6ae0ac260997c22270ee4fcaf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.40234375, -2.66796875, -3.03125, 1.1171875], "student_probs": [0.5615326166152954, 0.009586513973772526, 0.006666374392807484, 0.422214537858963], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a92989d8aa9b06fd955a92d755867198e094ab91441b6f0c9e79fc3b77a41450:action", "state_id": "34eb5382151cd0dc526ab60421369bd90ccddad4e6f47055f1450c67a5986961", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.33203125, -2.5234375, -2.97265625, 1.15234375], "student_probs": [0.5347021222114563, 0.011316264048218727, 0.007221208419650793, 0.446760356426239], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0e545a499e148c91101dc3fb3de6280bd4d22d60a8eb7fbbc2d7b68f818fbaa1:action", "state_id": "79db578c5cb88fad3113594f492da440615017a812d4f39ddedfaf009b802b30", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.5859375, -2.55859375, -3.25390625, 0.89453125], "student_probs": [0.6558966040611267, 0.01039652805775404, 0.005187020171433687, 0.328519731760025], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "631ea1fb040404f3c86bc5cba51ad36777d6dff744bf578a42c7074e05d7b69f:action", "state_id": "27ff2f35f4e11840e6fea0edf7076d835a97f4133a4be6fc5f4a658e78d0ef22", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.48046875, -2.5390625, -3.3828125, 0.53515625], "student_probs": [0.707091212272644, 0.012700336053967476, 0.005462346598505974, 0.27474603056907654], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "72ee7c0629b718f0c71c5eebdcf8560b31a695537a7b7bad464d12b91425be6f:action", "state_id": "8b14262b36058a5ed6b11a3e0465f746eef3b34e2e7a29039de8dfa1a22e2037", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.42578125, -2.1875, -3.06640625, 0.65234375], "student_probs": [0.6668517589569092, 0.017980476841330528, 0.007466156035661697, 0.3077015280723572], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a3b0f29258b00bb9f5addab295d00c5861be7ab70f729b0b0c260df56023c2b5:action", "state_id": "25d9ec71901cc9cde43ebadf49a186f18a26fc7ca50afe12baf16cf026d04737", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3828125, -2.2890625, -3.2265625, 0.47265625], "student_probs": [0.6954837441444397, 0.01768527925014496, 0.00692565506324172, 0.2799053192138672], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b1200c25fd8f9250edd09137f97634b3fcd45e835e964dfb9c3dd27fa691b36a:action", "state_id": "e496de1313faeec203d273480b3df103cd50e78464f4fc2e58a45616c2867fe3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.19921875, -2.05078125, -3.0078125, 0.578125], "student_probs": [0.6285271644592285, 0.024370642378926277, 0.009359089657664299, 0.3377430737018585], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f2c1bcccf7a0abeed2ca255e66301bfb5f1010cffcf41d4242b334e037444421:action", "state_id": "7911a9ff261191d5e0107cf3e92f129d26dec332df4b1d9d5b99d229efa66d45", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.078125, -1.3984375, -2.2421875, 1.525390625], "student_probs": [0.3725500702857971, 0.03130597248673439, 0.013464532792568207, 0.5826793909072876], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b28a964117546ea9bb6d7d64cfcf028f43b22d880c314000cfa024e6c05c98d4:action", "state_id": "45ee2fde66949fe8ce7dfe1e77624d3a67eb7099599384c540b765dad92ee7d5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.236328125, -1.8671875, -2.04296875, 1.669921875], "student_probs": [0.38090336322784424, 0.01709917187690735, 0.014342810027301311, 0.5876546502113342], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d34375035f8fa4820ac44658892d1ca3a5dcc9177497f57af1e0e003eca9f72f:action", "state_id": "1eb17bb9a8540c4ec919c46f271fe4ac4f94333ea1bac4143eea30fcefd1e34e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.146484375, -1.92578125, -2.21875, 1.541015625], "student_probs": [0.3899306356906891, 0.018060065805912018, 0.013473630882799625, 0.5785356163978577], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bd415806c58c69f4c024b9324421fd8990ae582fc7888f7522f9ef205d0f272d:action", "state_id": "04077637e732c9e8889eec8709709995d5e32dd240c4428c1adca5d5f3a869fa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3515625, -2.55859375, -3.1328125, 0.77734375], "student_probs": [0.6271691918373108, 0.012566820718348026, 0.007076938170939684, 0.3531869947910309], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "036444797bfbccd956f6ec8daedf21f0c9c8e19aa2408aae4111164e17554fe8:action", "state_id": "4edd9fcb7e1c415113da0659c0e9c4043f53e4769be45150ec472b27195cda06", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.2578125, -2.4609375, -3.10546875, 0.73046875], "student_probs": [0.6145634055137634, 0.014911938458681107, 0.007827403955161572, 0.36269718408584595], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1f6117fc13f26079bc6811fa071e61f034a2f62734e8cc4a68795cfaea0026cd:action", "state_id": "142ff1fcf1e764791d0be41a0b3c1a74e13aa55fb9e203dddb806bc392734cab", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.58984375, -2.43359375, -3.06640625, 0.96875], "student_probs": [0.6390798091888428, 0.01143400464206934, 0.006072554271668196, 0.3434136211872101], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "45dd0bf89e23341094aedc729f23a8edb15ad85d8ad15fc01a0b458bc6c8f0fe:action", "state_id": "10a953f384be6b3c37f870d849c62625688476091c59270f2d54ed3aa78f4ca6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.5234375, -2.46875, -3.16015625, 0.87890625], "student_probs": [0.6440752744674683, 0.011889172717928886, 0.005954944062978029, 0.33808061480522156], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e3adf42169038fb6826c7d16efa6103af36dd9d74e2e0687cf73b997dea53ca9:action", "state_id": "ffffb5aeb12c0315fcf037c430fac0ee98a09bea00455c1772f67ec24cb40275", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.37890625, -2.2578125, -3.01953125, 0.734375], "student_probs": [0.63957279920578, 0.016845468431711197, 0.007864531129598618, 0.33571723103523254], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f08a790b16afbe778a15b81935f805aeb5ffda6635a0374c56439a62e15538fd:action", "state_id": "d193dafeab0660852231d4c0c77a65dad486da66b295cc07d265cb62a78f53aa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.21875, -1.5, -2.23828125, 1.57421875], "student_probs": [0.3961448073387146, 0.026128580793738365, 0.012487756088376045, 0.5652389526367188], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2d71018f220a27a63af1c0b745eab816edf44a6a2405a2b987390ed62f8eafa5:action", "state_id": "32bc2731a20dc05d9939bef1cc4abe5859d35c76b53d7e1ae8f1542cec4c8273", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.32421875, -2.2578125, 3.5859375, 0.27734375], "student_probs": [0.035556595772504807, 0.00268879858776927, 0.9278262257575989, 0.033928342163562775], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7d658eb8be47bbd38cddb5852a5acee9275d49b63244118a984f98ce6479b9b2:action", "state_id": "652da2ecc8557cbf79302dec4cc7595c9e896f5372bbd4c5e0e1ea06dd494891", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.38671875, -2.01171875, 3.470703125, 0.3671875], "student_probs": [0.041811637580394745, 0.0037989974953234196, 0.9133864641189575, 0.04100292548537254], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "42d8c1f5f0ea6d0475fb504224b1e0576442baeda7f323342149e10993009109:action", "state_id": "9222f382f3e991c39ea9f05a13e701711600fb750fd73b45f20fe640e17bf600", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.52734375, -1.83984375, 3.513671875, 0.51953125], "student_probs": [0.04566468670964241, 0.004280791152268648, 0.9047452807426453, 0.04530932009220123], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "68a4026b7b1e211848960840e4aaad03cf626648c7a872749a4f56d007282305:action", "state_id": "42c0f3a24d286c6b4093e81403598eeab6b9653c4e30fc621f1659782535bbe8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.705078125, -1.669921875, 3.462890625, 0.712890625], "student_probs": [0.05597168207168579, 0.005206177476793528, 0.8824114203453064, 0.05641067400574684], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "318d3f0ae09d0f83f3830b41cdb859a36933296c0809a67427124482cb86157a:action", "state_id": "1d693267d80da2e8e6076eb9953aebc436b6ae755ddddbec749560d2716e713e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.607421875, -1.109375, 3.119140625, 1.849609375], "student_probs": [0.14546222984790802, 0.009613031521439552, 0.659601092338562, 0.18532370030879974], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "38e4a17bcdc93a27181923ce0bf5104a8ba17057832cdecf6a1199397a2c4c12:action", "state_id": "0d543a1896ac217a38db3243715acf0df2355a7dbe87a914557072143ddf8c7a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.1484375, -1.18359375, -1.9296875, 1.646484375], "student_probs": [0.35859668254852295, 0.034819137305021286, 0.016511768102645874, 0.5900723934173584], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3c439a27a9fe97e68209c8468aa419d662ee623de197db4933fa64ce429337e7:action", "state_id": "8d05a374e54a041aad84f5cbe134043cb506db58a15aff44310afb30187ea41f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.15625, -1.076171875, -1.98046875, 1.58203125], "student_probs": [0.3729284405708313, 0.04000341147184372, 0.016194438561797142, 0.5708736181259155], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c23bffd2eab5b59ae6955e4928684427733f1282c604a636864efbfe7c33777d:action", "state_id": "75a2ca84870ba1e575d7f0fd73550c1128ee6b251decb8c15f4f7e5d0fae3f36", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.14453125, -1.03515625, -2.07421875, 1.5078125], "student_probs": [0.38593316078186035, 0.04364011064171791, 0.015439270064234734, 0.5549874305725098], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6bd871b85e572312f9b42ffe63c1b6fd4f3e13d8fd3aca634cc8486e016a0a8e:action", "state_id": "b2569045207573ba2998a9a6c258ea1e4599a709526c746d920935a1cc6a2214", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.04296875, -2.51171875, 3.546875, -0.17578125], "student_probs": [0.02618573047220707, 0.0022176867350935936, 0.9486675262451172, 0.0229289922863245], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8334da0b4eb3c1142928d7a10d27ca64681a6c3296693e606cc33af6b0a3386a:action", "state_id": "e9a8a5b64a56891808c2a90433a97170d8117a29b9b6e1c3f337ba93f28ac817", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.09375, -2.2578125, 3.4609375, -0.025390625], "student_probs": [0.03227913752198219, 0.0030736280605196953, 0.9359936118125916, 0.028653642162680626], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "131c52cafe7d2ecb2931bd8841f76035ecf47f4ee8c19631f53c0018de111312:action", "state_id": "8c5c19249af3ecfc009dbf0056dff105757e51a8b9bdaf58766455db534260a7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.09375, -2.1171875, 3.529296875, 0.02734375], "student_probs": [0.03021719492971897, 0.0033117400016635656, 0.9381952285766602, 0.028275759890675545], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "843b3b2ead05cc5efa84f5511a2889062b36191dbcf4eb2aad8b8f79ef3941b9:action", "state_id": "8f3b01f2987180b545ca2e8e9f9ade9375ea6241d6d90a8ac6479bd57d6e4322", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.365234375, -1.943359375, 3.4765625, 0.212890625], "student_probs": [0.040968604385852814, 0.004072317387908697, 0.9197795987129211, 0.03517945110797882], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ab3129a7bcc15cc7b7c7fc459d0e45fd5c0fd694a012d9fc5fdbec57a3fff72f:action", "state_id": "ef942c3f4cd0031a96076f83474c272f979b34da779313c5cd30f67d08943801", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.630859375, -1.595703125, 3.51953125, 0.669921875], "student_probs": [0.04970879852771759, 0.005363514646887779, 0.8932386636734009, 0.051688969135284424], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f4f46bb84e927c6412fe94391248209d8b71027375b5243351b5fa1664bf9afb:action", "state_id": "32ad02021da74657042f9995d2bd347ebcd2604d4e761281f1e25e77e5a2643a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.603515625, -1.68359375, 3.51171875, 0.564453125], "student_probs": [0.04905064404010773, 0.004981564357876778, 0.8987963199615479, 0.047171544283628464], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "433f89aeb529c7c9439b4ea2489c1f19c6b2c96027b1356c5a169ee70644271e:action", "state_id": "17948cb77ad87200d8950f8df8e4dd85e5556b37f7f93190ba92c7ba735d61ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.603515625, -1.140625, 3.115234375, 1.828125], "student_probs": [0.14597144722938538, 0.00938647985458374, 0.6619101166725159, 0.1827319711446762], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dda56b4ace6e5304826b9a4c5ed459a23109494e9ec4b681706bf852e7063720:action", "state_id": "d861ad8c9f679de10a00f2da4785088c92b87e8a41c6950d002fe46ef2cf86a1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.173828125, -0.947265625, -1.87109375, 1.759765625], "student_probs": [0.33736199140548706, 0.0404498428106308, 0.016058441251516342, 0.6061297059059143], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f020e9fe2fca4df9ef2a4cf555c75247c10a522ecb5197e98ed51e8a91d57a5a:action", "state_id": "01098b862b17480070304047f6910c03c63c2d68b6f8566659233fe610de2975", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.0625, -2.52734375, 3.578125, -0.203125], "student_probs": [0.02495664544403553, 0.0021218671463429928, 0.9512388110160828, 0.021682709455490112], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "745c5b4b2ad1731a2b8f4fc4537e7d5f3ab9fde068e1480c53f67c9a628a42a6:action", "state_id": "7f1e17fc8de14bb4e0a1dd0a6262e676a62e466b5adb2c036468ae33a3dea20f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.09375, -2.2734375, 3.4765625, -0.0390625], "student_probs": [0.03182395547628403, 0.0029833053704351187, 0.9373266696929932, 0.02786598913371563], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ef5e3d0e0c71c49e636bf2d417257c479e890eaada4ca625538758c1450cc43c:action", "state_id": "2ec2e7191cf5c08fed546012bdbbf1531f119f345aea8f9896390d9a98feee4d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.23046875, -2.0703125, 3.515625, 0.10546875], "student_probs": [0.03484826162457466, 0.003491118084639311, 0.930907130241394, 0.03075348399579525], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e0aa062bca57968b4be8037b547f13dda8dfa2606ee5ee2126346076718f1ddc:action", "state_id": "7bda4ab547fbdd50c0b8412de4210f9de6b840895e3e9fb19b98373c87276a74", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.6171875, -1.78125, 3.44140625, 0.671875], "student_probs": [0.05264585465192795, 0.004783392418175936, 0.8869656324386597, 0.05560510233044624], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e085ef02ea90eb164e7d94a306f156d4b38defbc100e08f625f6655dc5766374:action", "state_id": "aa12679e9916ba72d94006ce1003530de750a2d9c24897da2ef486c5dc7fe55b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.611328125, -1.70703125, 3.52734375, 0.611328125], "student_probs": [0.048623956739902496, 0.004786296747624874, 0.8979657888412476, 0.048623956739902496], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4066ce76183579306cab83703a3264f90d91ec612cc5ca9d4db2d9e25a59bf5f:action", "state_id": "c7901cfd4cb7aeaf1a5d31df6573f00100027f553f01b85af51969cef72d961c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.61328125, -1.642578125, 3.515625, 0.58984375], "student_probs": [0.04926494136452675, 0.00516215106472373, 0.8974491357803345, 0.04812372103333473], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f3ed55cd1f065e777c00718bf6bc0b21e55df2920706dde5672bc1b4d6229d03:action", "state_id": "6e2dd7e3003ad9dd0f5f7f20fce830c07409881f8cd679f4fa89c0a20325864c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.62109375, -1.12109375, 3.1171875, 1.830078125], "student_probs": [0.14790555834770203, 0.009529444389045238, 0.6602824330329895, 0.18228261172771454], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "79fa594e2f2df76e02f13bb4120dd7cb3f2d8e07445bfab7ec28e6d352031f66:action", "state_id": "31ce9b9d8ca9cdb085abc7d8691386b33e47ecbf7f9b349eb5a07894b7c0069c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.20703125, -1.033203125, -1.86328125, 1.76171875], "student_probs": [0.3455142080783844, 0.03677430748939514, 0.01603415608406067, 0.6016772985458374], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6733e282f55d590589e1be18f2ea69a945dcdb6925f0cc5e232530095636c4fe:action", "state_id": "8cec9acb545da51685918abc634fc9f344a2bb444109a95346b46af0a5214ea9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.40625, -2.2109375, 3.556640625, 0.37109375], "student_probs": [0.039395447820425034, 0.0028761792927980423, 0.9196938872337341, 0.03803451359272003], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6556e9ebbd1aaaf6b8cb11faf87b79c1efb60ba068fff50981702eaa29cf0907:action", "state_id": "fae0bcb1733a2ee9121c2068649d80506531c9361de6a99ae8380c9f9a7388ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.41796875, -1.9609375, 3.45703125, 0.41015625], "student_probs": [0.0435340516269207, 0.004033511038869619, 0.9092371463775635, 0.04319526627659798], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "55cde20efe03e73740d471776e4e26d8e92c99e20f964bf95733056b087b6e99:action", "state_id": "a0ed58d1d48877338422f25975bbd983d591accab34eee76b6045228e73a6d99", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.53515625, -1.82421875, 3.513671875, 0.50390625], "student_probs": [0.04603558033704758, 0.004349407274276018, 0.9049957394599915, 0.04461921751499176], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d558ea33f1c3fa7e259da6a63f921b41913180de00cf31c7ca0d6ee189a3f5f7:action", "state_id": "fba6f090106fa9b700efd4e572431a0b425f752334e202d807cb376012e36ba0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.544921875, -1.453125, 3.208984375, 1.625], "student_probs": [0.13488037884235382, 0.006728427018970251, 0.7122655510902405, 0.14612559974193573], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "3c38935f17b4035bd2009218a4c9bda3b9b307e269bbacaa36adcfce73059528:action", "state_id": "fe353286ff7f9cf414ad6b61e0b5a3ac27069169519a04990fb17bf0afc8cf32", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.138671875, -1.05859375, -1.875, 1.732421875], "student_probs": [0.33658885955810547, 0.037397224456071854, 0.016530221328139305, 0.6094836592674255], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a145c643f5b6f0fae42b3471c910e56d009395fc730c8f8aa3fe47aca2b8dfe0:action", "state_id": "90220bf826989b3c044f5f995c30fc942c4ef5e476f4e408e83bd98929e7018d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.1640625, -1.095703125, -1.98046875, 1.59765625], "student_probs": [0.3717121183872223, 0.03879743069410324, 0.016016002744436264, 0.5734744668006897], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6011b366c69a3fde6b5a74614ae12aca2ca443d1c56e2ba1eb3921ed99cbe9b2:action", "state_id": "280860a7e8a677c6552b1d61f1c0e48b3eb4ad7ebb2374ab9e21985f8b093c11", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3046875, -2.25, 3.5859375, 0.27734375], "student_probs": [0.03489213064312935, 0.002711694920435548, 0.9284451603889465, 0.03395097702741623], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "a07124d4af3507eabf963896cfa65c20d810991d1f479c4d097df539028a6d9d:action", "state_id": "e2e9fc2535c52320407488eb75fc4bf62137f13ad1e7ce4bbe5552e4599f9d8a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.37890625, -2.00390625, 3.470703125, 0.35546875], "student_probs": [0.04151836037635803, 0.003831756068393588, 0.9140932559967041, 0.040556587278842926], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d6f3d465dacc0a12ebe30e5269c7ffad6ed9e74f3448e3a8c0ec096a1dc7620c:action", "state_id": "7c6ca19ba6c2088e83a077e937f64025a0a20a5a23e2304213e1e5b7b114e40d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.568359375, -1.82421875, 3.513671875, 0.552734375], "student_probs": [0.047410231083631516, 0.004332998767495155, 0.9015815854072571, 0.04667520150542259], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4c6a634781efdd4f9ede4491512ec04d01330e029f321c842a0debb97b837302:action", "state_id": "9cb32c8846d1f13f8ad25a2b014b3be0f92a1e23f62ce963e8c5f3a2cfd8f55c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.697265625, -1.677734375, 3.462890625, 0.712890625], "student_probs": [0.05556256324052811, 0.005168123636394739, 0.8828317523002625, 0.05643754452466965], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "907eaca04a852664ade482cb18ca36cb83711afbcc3c5480ac4ce08abc98431e:action", "state_id": "52b80f7b78fcfa536a561b1ad85733291ef0974a515eb87d65f66a432e8db5c4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.591796875, -1.11328125, 3.1015625, 1.849609375], "student_probs": [0.14520889520645142, 0.009709406644105911, 0.6571674942970276, 0.18791426718235016], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f62c98b39d466f03515fae7225ac20b5b8fc1fc74d02e07ecbaf1ee370c46d4a:action", "state_id": "7276bc2c098fa2e0e80f683e6c396f04d9960cf82b6de33d7bbabceff6e68b31", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.15625, -1.18359375, -1.94140625, 1.646484375], "student_probs": [0.36046475172042847, 0.034728143364191055, 0.016276752576231956, 0.5885303616523743], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d9f768f8605f4aafb17cc86804f5cbf44667d8784ec29052e806b82b1dd0ac00:action", "state_id": "5732be6cfae925bbb5e13ff26e5719811728cf76d2af4096e9f86d9ab97b47fb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.36328125, -2.4140625, -2.97265625, 0.859375], "student_probs": [0.6097043752670288, 0.013952106237411499, 0.00798078440129757, 0.3683626651763916], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "59f2726b7f52f81dc00bd89a9d25971ba99db7aab7ac3cbfcd5fca00d8834e87:action", "state_id": "a86bac1a85cffd191202cba9eb6a739b24932871896620e9bbaa2dfec2b99d4f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.119140625, -1.91796875, -2.42578125, 1.427734375], "student_probs": [0.41011282801628113, 0.01967449113726616, 0.011840317398309708, 0.5583723783493042], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d3fd84d76d66412b658769e57720a6beaa0a1f57fdc91940aa98ac1938d5cf64:action", "state_id": "7b636bad667bb63046b7cbf2e822954346c9e33d2e27614f12e6ec07c9d6fb82", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.462890625, -1.87890625, -2.21484375, 1.69921875], "student_probs": [0.42969292402267456, 0.015199673362076283, 0.010862717404961586, 0.5442447066307068], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "41fd54780082bd534be8e680fac3389f00fed6fb347d3b1b2a92679d412ff4be:action", "state_id": "7322bdeeaebdca739710e947f459d8aa21984da360a29fefd3da732bb8de447b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.345703125, -1.8203125, -2.29296875, 1.58984375], "student_probs": [0.426442414522171, 0.017983626574277878, 0.01120999176055193, 0.5443639159202576], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b60ff8b00a8fbdd1728bb2421ec36a50c552231ce1361095ab487436476429fd:action", "state_id": "48563e5513421536ad49c5079b22867432b9119aa7e2a4bcd545fb70f2ca1465", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.39453125, -2.6875, -3.03125, 1.09375], "student_probs": [0.5652662515640259, 0.009537826292216778, 0.006763331592082977, 0.4184325039386749], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e31af7519e868e957566cfe39fcd35639df60036d1a8b97de0e9b2dcbdd9929d:action", "state_id": "5ab27abdc325d3540de87f35b10e439afd1a42fd9e69c1d353048cd0c779780d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.31640625, -2.59765625, -3.03515625, 1.078125], "student_probs": [0.5492010116577148, 0.010961641557514668, 0.007077367510646582, 0.43275997042655945], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8759eb489f6820b142b440c0f3ec42323c60e6ddd4d68f8b66b206474f2971d6:action", "state_id": "f31c3a74a2f686988c54f60f2ed6d64406feb30360eea1ff4621c37b27765ce1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.5859375, -2.53515625, -3.26953125, 0.875], "student_probs": [0.6599806547164917, 0.010709346272051334, 0.005138400010764599, 0.32417160272598267], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ceddb79b6e4598f62720e1cc4b01caef9fedb0b966a46e16c67a0f8141d1953d:action", "state_id": "efa50a406acbbffa0765bb08a2c11e3b877d9e181eb4d60e45808c5f64e167e1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.5078125, -2.5234375, -3.3046875, 0.703125], "student_probs": [0.6788371801376343, 0.012050802819430828, 0.005517259705811739, 0.30359476804733276], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4c4d12b1ed0594e1f0e2afbccfabee381e0f128163ca54b3755f1b8a6a721059:action", "state_id": "444dc9d0432981b7e895842d0b4297ef170a26714590d39c2fb6ba65063febee", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.40234375, -2.23046875, -3.01953125, 0.734375], "student_probs": [0.6446612477302551, 0.017045946791768074, 0.007743470370769501, 0.330549418926239], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6102d67d726079de8e1221b3d22f812b53e608ddc1bb5ad47f9e24cf180858ac:action", "state_id": "d7742f28e8a584b157217e2b6f90f20278ab3e2edfb000161ce82f0e354f05c5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3984375, -2.30859375, -3.2578125, 0.453125], "student_probs": [0.7029330730438232, 0.017257217317819595, 0.006679289974272251, 0.2731303572654724], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "05b622684f668b9a620015c6180ea82f3be7f5a86becfbcaccd9d1a954ff4411:action", "state_id": "d31cf31f0e9f7d652ada1ecb381eb932effbff2f61eb52b940c089b85b8b796f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.19140625, -2.09765625, -2.9921875, 0.609375], "student_probs": [0.6206189393997192, 0.0231421310454607, 0.00946048367768526, 0.34677842259407043], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ba09d5ae63483f639be2e76c16e4191036b6b9b5c5077ba6d3029caf7d4712aa:action", "state_id": "eddc2b4450640ccfe0bc2ed607cb585ff9c80d3da398720230bbc9d07f9ca9f8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.037109375, -1.453125, -2.2734375, 1.494140625], "student_probs": [0.3705398738384247, 0.03071424923837185, 0.013523301109671593, 0.585222601890564], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "8f92072b88e027bbee22f779acffa613400af0f95a5f2f849d37d5c442b2fb03:action", "state_id": "6ea7489e8002c4128bde766d3e45ef00c493576ec5b5aaf923c1b4d07fac9cf0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.07421875, -1.5625, -2.12890625, 1.5078125], "student_probs": [0.3766446113586426, 0.026966173201799393, 0.015304960310459137, 0.5810842514038086], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4fce44ec01dfea0f8bb4aca29df9e57403e70fafbe2e6379a8f22e12aa307495:action", "state_id": "c20ef5bf8ef6e59f6a13c789c7748fe497618f384fa9e0e0ea2426b6c5f3035d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3828125, -2.2578125, 3.5546875, 0.328125], "student_probs": [0.038654424250125885, 0.0027567052748054266, 0.9219916462898254, 0.036597270518541336], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "423e9adb0b2997c3b62e2aa556df86913ecabb6cd3b3cb33097c62c46a10ba4f:action", "state_id": "5d016a082bbdb829523b0fe2a684c4002dc749a51a37ea79784be96fdd0f0d09", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.4375, -1.9296875, 3.45703125, 0.4296875], "student_probs": [0.04431121051311493, 0.0041539110243320465, 0.9075684547424316, 0.0439663790166378], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "86cb3da9745f358cf9717cfefe95039c01021110611a2784dcf652ba9412b65d:action", "state_id": "487a2365645c647845ae02cfb0398ded98af4c1c94fd5981a2cbae47b8b27e7d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.44921875, -1.625, 3.25390625, 1.369140625], "student_probs": [0.12426464259624481, 0.005744223482906818, 0.7552893757820129, 0.1147017553448677], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c03e77d00e365f0f292aa96ebc80cb1af5f4d0846296f2ee8f47671ddad4e7a2:action", "state_id": "df5b51ed9f1d4e9fd21f502e8713e4cee542844d3aa9ceb1d9b73a2aea1a0acf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.546875, -1.43359375, 3.1953125, 1.626953125], "student_probs": [0.1363700032234192, 0.006923372391611338, 0.7089672088623047, 0.14773939549922943], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ea7d8b515e3b1eb6a0caf11c1207e0945cc0052e8ea725fee066c2e010a0bcb5:action", "state_id": "64f125ec3db15db02ab8a6a462d4e2e55d1616411a9385cea4189bde54d165b2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.169921875, -0.998046875, -1.875, 1.763671875], "student_probs": [0.33638596534729004, 0.03848583996295929, 0.016011981293559074, 0.60911625623703], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e703826cb0cc7f3eecb77fd56c3385538ad39ff943274889d24a6aa1fb45beb4:action", "state_id": "c7a120c5511d70f896e14882c3388f6a9f006baa3e7069d76aee188b39055588", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.2109375, -1.953125, -2.1328125, 1.58984375], "student_probs": [0.39397314190864563, 0.016646837815642357, 0.01390895340591669, 0.5754711031913757], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2c8c07d632548a7bd1543e6f42363b5f19491f30326e768bbc49f69e4e38a511:action", "state_id": "30728eda700ad9b000c79eb4c0085eaa30ae638e8333a2eb4c949cbde46ceb1d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.11328125, -2.01171875, -2.34765625, 1.4609375], "student_probs": [0.4014318585395813, 0.01763768494129181, 0.012605085037648678, 0.5683253407478333], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7b1fcf103c1aa53922a0e406666c25753594bb6f0b43a6078b292d94eb88a34f:action", "state_id": "e892618ae8f97a626b18b6d2fb9d2d134cefe5f5bf6f9e041ce316eeeb9de156", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3671875, -2.7109375, -3.17578125, 0.73828125], "student_probs": [0.6407153010368347, 0.01085320208221674, 0.006818342953920364, 0.34161314368247986], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "15638056b9120b3d5c15c8d8e7757e9954c947d7a0dbf8e4ad9f2b2e5f838d96:action", "state_id": "9efe780edb5f21776454d1db8ad8481c2cc5c63fba16b54360e13ce60c9af927", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.27734375, -2.69140625, -3.35546875, 0.49609375], "student_probs": [0.6727402210235596, 0.01271279901266098, 0.006543987896293402, 0.30800291895866394], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "65c46289ae25f3cb652e7fe22af095c5efb27330c568a9eef658c994a2359f81:action", "state_id": "05abb2d3f4f709047f18de989743bcc78911aac549dac852c0bf790bdd68f888", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.5703125, -2.58203125, -3.3671875, 0.875], "student_probs": [0.6571086049079895, 0.010334683582186699, 0.004713116213679314, 0.32784363627433777], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f2c10466ce6f4a5826a03e6459d86e4cff87599a84fca61b5a0f9750c68ca2ef:action", "state_id": "3c8a1e6ff79e645ff6e513cc31e6a493986a960409f0513df9e1251836823304", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.50390625, -2.5390625, -3.4140625, 0.48046875], "student_probs": [0.7224301695823669, 0.01267525926232338, 0.005283834412693977, 0.25961071252822876], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "39967bfbf62be32c4893e26352a9f2385150c5934205ee0260e5cc51a82da611:action", "state_id": "ff7502bfd511f2277d3f7cbff27c34bbb502c01e82a5ffda49ee180d4545857d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.4140625, -2.18359375, -2.94140625, 0.80078125], "student_probs": [0.6321930885314941, 0.017314400523900986, 0.00811509694904089, 0.3423773944377899], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "eb0c84c488dab99edc4a52b5252652270c30cbd58a21f838ef19bbb5f3403c2d:action", "state_id": "92d1bc3f1ad27136ac39225d3375c03d98aec96eaedd562b04e2b5ef6f59d7ef", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.3359375, -2.27734375, -3.2734375, 0.40234375], "student_probs": [0.699271559715271, 0.018854618072509766, 0.006963374093174934, 0.27491044998168945], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "5b9d74a62cf928e6e2ce81f4b93405a8aef98438904da411751edb4feb86490c:action", "state_id": "44922748e102c0bc77b6c257de4341d64c203bda345bffab1cf6d99ae6e9d589", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.0703125, -1.41796875, -2.22265625, 1.54296875], "student_probs": [0.36704713106155396, 0.030484214425086975, 0.013633384369313717, 0.5888352990150452], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "daa2958ee08891078a7dff2fe41862772ffaaea604be6b2f9a24aead605da5e1:action", "state_id": "41bd9de6bb45157015a6c08baba7d966a749ebeae8861220be7fb1855e6e9486", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.236328125, -1.8515625, -2.00390625, 1.703125], "student_probs": [0.37318581342697144, 0.017016541212797165, 0.014611984603106976, 0.5951856374740601], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2aba63c7851c4c770a5096ffe9daaae982a1b602f25d4a9e9748a5e655acaa35:action", "state_id": "6f4990036de1afab2156f2ec492e94764ea123863e868efe4957c50c64ed28a3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.171875, -1.8515625, -2.12890625, 1.62109375], "student_probs": [0.3769921064376831, 0.018334541469812393, 0.013893804512917995, 0.5907796025276184], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dfa2739eb3ee2ac606546f0e63e47e7beae60cc33d30e2237d922933ffed2463:action", "state_id": "4e69beeee2bfa9c2c08e4b6b5cd388025b2e283b46cafca51d5c700834eb5350", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.29296875, -2.25, 3.587890625, 0.26953125], "student_probs": [0.03444620594382286, 0.00270859501324594, 0.9291969537734985, 0.033648259937763214], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "be3f35e0b30839b3e46428c6d7ab0fdf0904acab7f040215f968b8c0491deab1:action", "state_id": "0c473ecea5b207fbe801781cae9203f2f28beb2002be44b58fae74018c537833", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.546875, -2.12890625, 3.37109375, 0.6171875], "student_probs": [0.052660755813121796, 0.003625851823017001, 0.8872166872024536, 0.056496743112802505], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "e2b8377c122ee39e7df8850864cd76b732c7e8c09ad9bd3af21d5a15cd94acf7:action", "state_id": "0da2ca5b323af025aeba3aff0c8b93c2921da431f7e2971a1ac6ff448e571861", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3671875, -2.08984375, 3.484375, 0.33984375], "student_probs": [0.04058196023106575, 0.0034774260129779577, 0.9164533019065857, 0.0394873321056366], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "662717de0e2a1ad43b51edeedeb73f0e2e520a94f1bdc98d93c20af83c1ec132:action", "state_id": "07c2ad8f5dc68bb73e30127661b93512070d0317440c761b6ced2717459f8cfd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.337890625, -1.974609375, 3.5234375, 0.357421875], "student_probs": [0.03802391141653061, 0.0037648770958185196, 0.9194374680519104, 0.03877386450767517], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "2b7250470292acb6a5d1c180dfdaa210d31d0581e51215e6b63b5fa93abfebfa:action", "state_id": "20202db43ff4eed7dbc88b40c0a2c3258c0ca148b080653655b71975f04ac919", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.76171875, -1.638671875, 3.46875, 0.826171875], "student_probs": [0.05833631753921509, 0.0052900840528309345, 0.8741535544395447, 0.06222008913755417], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cc5c1825dbd0a249eb3f8410252737e8cfea3e429ebf9c6f141247170fbf03d0:action", "state_id": "4006341fb21c6cd2f9d1381ba051dc097a3ec7cb88694211cd23b1809f1c997e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.64453125, -1.619140625, 3.5, 0.65234375], "student_probs": [0.05129679664969444, 0.005333226639777422, 0.891670823097229, 0.05169912055134773], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dae85ea13e3f82b84e8b3c5aa9ce5e36934f4e7c6483d5a98b72ab337bf0af79:action", "state_id": "9aa511814531c89080ee2826c517302c7f671e3fa3632b7cfa8326d6d3dd737d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.169921875, -1.09375, -1.8515625, 1.732421875], "student_probs": [0.3439083397388458, 0.035755470395088196, 0.016758251935243607, 0.6035779118537903], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "861c00bc40d676bc2d2a45ed5b5c658b53426d78f43d385bdec6598a55aa7015:action", "state_id": "b4bc1046919259913b9d8fbb374ec9ed898ef0485c331d65ee8e66f6327bef1f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.166015625, -1.05078125, -1.9375, 1.6640625], "student_probs": [0.3572254180908203, 0.03892240673303604, 0.016036244109272957, 0.5878159403800964], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "0a64aa7e273ed6a7104d4275e974018bad372a3dd36ce1e03f7f45a7239fbfe6:action", "state_id": "4f3c64cd0f120ffc792672bc23dadbdc1a8cbae7762f98da9947cc23e3e27140", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.51171875, -1.99609375, -2.00390625, 1.60546875], "student_probs": [0.4633970260620117, 0.013884480111300945, 0.013776430860161781, 0.5089420676231384], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1bd383a52a73f4b3e83a81ee422f9554265b589f69058375aed6335957b9f3c4:action", "state_id": "5d9f85dc16b0bd78833581992d86d69ce733776ad2b1dace391d6f21d628f398", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.52734375, -1.92578125, -2.01953125, 1.623046875], "student_probs": [0.46277156472206116, 0.014645140618085861, 0.013334551826119423, 0.5092487931251526], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "1454cb0c28ec48f986569bedece8bf88548be2031ec2c058ac48e94727fe3c26:action", "state_id": "2d40aa1da01fd00d492855b87462aea16c74c09630392c196f1bae529e32b97c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.609375, -2.3984375, -3.0, 1.04296875], "student_probs": [0.6266871094703674, 0.011388851329684258, 0.006240575108677149, 0.35568344593048096], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "06a5a4eb0d9b9f000ea8f95a6e2769d960bbd676ce4f746fecbca2a59ed97d5c:action", "state_id": "b29097e377cf4403a3d08f8ab7c220c39adc594d7333617b6baa92318d5968a1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.5078125, -2.4921875, -3.2734375, 0.71875], "student_probs": [0.6752323508262634, 0.012367311865091324, 0.00566216791048646, 0.30673813819885254], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c5a8ad33856b2e6bdf65dd65c6ec456723f0ebc0e95852781c62fb0af6c04cdf:action", "state_id": "a27c3c828fd5eb00474af8f65430b48e8c579c96132730eb40a5843b6eb42b50", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.41796875, -2.203125, -3.00390625, 0.7578125], "student_probs": [0.6428653597831726, 0.0171988345682621, 0.007721899077296257, 0.3322139084339142], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "dc79b45ff4aa8a88a5104ba6da4950818f7f82e42bb63b5ffe4381eecbd79750:action", "state_id": "7ed2f2693d57972ea6f572dafbaa0ccbf460d94025bd2ef74e8f39d6edbcce91", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.37890625, -2.28125, -3.3046875, 0.390625], "student_probs": [0.7106360197067261, 0.01828359253704548, 0.006570346653461456, 0.26451000571250916], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "cc634cf3fe799ec877c299423254ca76832081a99b50ffb13af084ce366e3fc1:action", "state_id": "91a17bb32fed8f98ed3a8be33043b410a711e312e0365a7fe584b094fb8ee634", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.23046875, -2.05078125, -2.9453125, 0.640625], "student_probs": [0.6221387982368469, 0.02338075451552868, 0.00955803319811821, 0.34492239356040955], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "7eb065868aee761d29175813449d394f48acc47b574897df4eaaad6a3f16c4bf:action", "state_id": "c7cce27ec3467525207010cc037223724dcabf6eaf06faee926a8158cd3272ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.021484375, -1.50390625, -2.3203125, 1.462890625], "student_probs": [0.37449076771736145, 0.029969388619065285, 0.013246987946331501, 0.5822927951812744], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6b3e68133b52f80330644c2c40b0b8cce784ae9fb62d17d8511f5ec2e024c0ca:action", "state_id": "9fc0d70c69e3ed644fee4e4cebd556d8fa6d1bb7a3d91309d1fb4cd8ec7de099", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.041015625, -1.6015625, -2.16796875, 1.482421875], "student_probs": [0.3750280439853668, 0.026693569496273994, 0.015150240622460842, 0.5831282138824463], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "ed9dff1fddbbe499bd2ed28bc52a457a6ff6fa571f7172e4d7c337d75d51c163:action", "state_id": "c193c304487c30459fff62d988ea0e96a31c091b98db21e08749117ed62ae821", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.26171875, -2.2734375, 3.5859375, 0.23828125], "student_probs": [0.033519502729177475, 0.0026563983410596848, 0.9310809969902039, 0.03274302929639816], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4ea6f5ad0e24bafa81622ee426d41671a41c8010cc23307c52a96cfb4d954bd1:action", "state_id": "67f47e99c33053abf7a49239ff68490bb864a8014f6c6d7f3c1b3dcb5e3ac60e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.3359375, -2.07421875, 3.453125, 0.30859375], "student_probs": [0.040575187653303146, 0.0036437029484659433, 0.916300356388092, 0.03948074206709862], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "61b50f8f4fda40bf50ce978f7e9b2bd2d4361bae09545f4269512e7073b15d76:action", "state_id": "17b395d1b33b0364e98711fef350e3bc509f2e1debbc93d154dd4ae0008e0e0a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.341796875, -1.943359375, 3.509765625, 0.330078125], "student_probs": [0.038685791194438934, 0.003936594817787409, 0.919142484664917, 0.03823509067296982], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "708ef8aff25b7eb782008553fdbe6405b9445bbb5ddfbbd3a5f3d724ba9f421e:action", "state_id": "8da134fcd4059d2347c136a7d60a6819fe6a59e302fb984715db530e660ebba5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.673828125, -1.69921875, 3.4765625, 0.689453125], "student_probs": [0.05376743525266647, 0.005010928027331829, 0.8866074681282043, 0.054614149034023285], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "c100318a36e3a245e196f2ab0930b9a5c2da609793073159b86287b829a4587e:action", "state_id": "60591ad4382476b8b3168dc7ec825f74ef8b18c7cbb842d2c47ca05628a7671f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.58984375, -1.125, 3.1015625, 1.83203125], "student_probs": [0.14545948803424835, 0.009631643071770668, 0.659588634967804, 0.1853201985359192], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "19fa683292905c991721f404d33f2f7c59b0a84365d889906d4d64150753472a:action", "state_id": "a175b1c0cf3115c1dbbd7e990574a612ee5444357cd2cbb988ada4b58d824915", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.158203125, -1.1953125, -1.8984375, 1.6484375], "student_probs": [0.3603891134262085, 0.03424938768148422, 0.016954675316810608, 0.5884068608283997], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "888b533734e6964d797fcd2fcbaa5b4b07236c7cf91b549439d8088d99a05b30:action", "state_id": "adcd8f9b8f6a6499f155405b41b66f3442f196403757744d61bebb165babce4e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.220703125, -1.8515625, -2.01171875, 1.703125], "student_probs": [0.3695804178714752, 0.01711752451956272, 0.01458431314677, 0.5987177491188049], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "4247ba80a06377efb75cd7391e14d10de334c3d0d0d9af99b5ad59c544c90456:action", "state_id": "bb5952be7b150013cd3469376088f545ebe755a218e6cf45156a4a2802945f7d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.171875, -1.8359375, -2.140625, 1.62109375], "student_probs": [0.37694427371025085, 0.01862090453505516, 0.013730193488299847, 0.590704619884491], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "f8ac5562f53e7ca9db9e78e2f5864c91ab4c3f234ba49433f7697035af3f6c1b:action", "state_id": "04b09c0ed6902d3bcff39fec7d624552edb4cbfb7bc81938e555484703737944", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.220703125, -1.9296875, -2.08984375, 1.607421875], "student_probs": [0.3919302821159363, 0.016788486391305923, 0.014303970150649548, 0.5769771933555603], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "21ca18c5b3554869c944eae3a0ab655ab88cb0c4ba9de43d98c0466d7e4bbe56:action", "state_id": "b65a00bf0ad15185986309272e688168328a697143f3df25d7f38b34a26c71d1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.130859375, -1.98046875, -2.2734375, 1.494140625], "student_probs": [0.39748600125312805, 0.01770472526550293, 0.01320852991193533, 0.5716008543968201], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "101f8e754c3dea5b65fe76d2c2acfbeac05cc4d1b7a537aa235a5035e83ec016:action", "state_id": "26bb2339b2c286176dc535a8780d854a3accc6adf77a74a0f00fa3c0bc74e7f5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.37109375, -2.47265625, -2.98828125, 0.90625], "student_probs": [0.6015280485153198, 0.01288061123341322, 0.007691363804042339, 0.3778999149799347], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "d0f6373bae2c3407f38bbf6ad9b7c4a82da822ba9a270f9459a0b821ac70dd27:action", "state_id": "eb9a54b13b619493f7a3252d7fbdb72bdaabc46bb6433878e9ca01653f71b78f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.265625, -2.4453125, -3.1015625, 0.73828125], "student_probs": [0.6145103573799133, 0.015027596615254879, 0.0077962144277989864, 0.36266589164733887], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b264ae8d48b897ffbe38c117ae8c5a484169884de525312be456721ddd1e7f8c:action", "state_id": "8327668d94e97e452c8107df0876ef828be9208b1279adddce5bb10dab130973", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.44140625, -2.40625, -2.97265625, 0.90625], "student_probs": [0.6176601648330688, 0.01317448727786541, 0.0074773309752345085, 0.3616880774497986], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "32f3aeb4f85e6d085ce7e50673fea3ea259925e9e1fa03f6b6870b4f45de8132:action", "state_id": "25bf6c0f1af769dc81cd8e0326d5e4cb46a1ab460daf8711489008c7343f2522", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [1.16796875, -2.0859375, -2.54296875, 0.41015625], "student_probs": [0.652840256690979, 0.025214677676558495, 0.015964938327670097, 0.3059800863265991], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "b643ef924c204d8d985159035095a637fd3bc515d38a1cec0f702dc5ed38b8e7:action", "state_id": "7904f49573877088ac21a6a2c7238c285c206e5a8fba034fec8c607a38240408", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.984375, -1.72265625, -2.1953125, 1.0546875], "student_probs": [0.45846816897392273, 0.030595703050494194, 0.019071657210588455, 0.49186453223228455], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "bc99e47de217fe40249bd59a09652808b3224ea45acf69a25e7f9383c0af07b0:action", "state_id": "a5f503c003da54b3848e2e6694da16ad15408a49a69caa9ac9f0b3852f1b7697", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [0.9453125, -1.53125, -1.921875, 1.0078125], "student_probs": [0.45343559980392456, 0.03810291364789009, 0.025781720876693726, 0.48267969489097595], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "fbd05c4135730416ca5bd1e313e14289c3c148ac667cbd1993ed404ed84ebcd0", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "basic", "target_transform": "expert_action_one_hot"} {"id": "6a3ffaebde1fd0e46b9fab85ea33f60ab2271faeca39c4c3dd1c6a2c4bb463aa:action", "state_id": "92f3227ef44c1f088a18ca156e6d8684b35c3d3474bd32d2a28ab355d48c5353", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.15234375, -1.296875, -0.471435546875, 0.728515625], "student_probs": [0.09615164250135422, 0.08321230858564377, 0.18996404111385345, 0.630672037601471], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a43e7d646c8b3338e35a79d3058010f2818979e23e827b19052d8c5533dc3432:action", "state_id": "0e954dd1656b3cad1585b530d159c3c8004f0f6d586c15fa5006b9971bacc028", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.984375, -0.982421875, -0.3427734375, 0.953125], "student_probs": [0.09222666919231415, 0.0924069732427597, 0.17518645524978638, 0.6401799321174622], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "788771109ff108a2c47d0b9baf568ba139c5a1f1bf18a52121ec3014c1136047:action", "state_id": "584fb2941abeb171dd6c7edcb6a8f1165da11944e06c3cc0d2150f7e7653cf44", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.07421875, -1.087890625, -0.3515625, 0.78125], "student_probs": [0.09577614068984985, 0.09447560459375381, 0.19728903472423553, 0.6124591827392578], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0167d10f053233b0beea21a5172e42d2e94a0e5eac9c2b18b6a9ba780fb83bac:action", "state_id": "37eb352fc6f1938dd09dc048590d6adc3f191fcc64abb9432f73e3ade4dcd8eb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.138671875, -1.01953125, -0.44384765625, 0.814453125], "student_probs": [0.08944086730480194, 0.1007576659321785, 0.17918197810649872, 0.6306195259094238], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8306c7c86bdd93644074d264b20131e0a5bdface9071137cabb108329ecf9a92:action", "state_id": "c393d67be01fcf09b810d567ac68788a127ff0bcacfbd73732d26dee90b4f9bc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.107421875, -1.109375, -0.6767578125, 0.85546875], "student_probs": [0.0938420221209526, 0.0936589166522026, 0.1443551927804947, 0.6681438088417053], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5ed94ee3f8146f18b36f77554949c1ba015ccde41bd342a1f080db9d5bad8d66:action", "state_id": "eb8cadcff8cbbd24d49da5e028d3118d81b7e83e2a9d3a611a41363a755d8118", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.146484375, -1.08203125, -0.555084228515625, 0.7734375], "student_probs": [0.09351459890604019, 0.0997403934597969, 0.1689356416463852, 0.6378093957901001], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a163e18c43a056e27d3482c9cd839781772a7207a1b84321671fd490c778a166:action", "state_id": "407c0a9808f9323a4bfed7d5e83ccc40067293c247a880569f6db05bc981087d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.24609375, -1.25390625, -0.83984375, 0.921875], "student_probs": [0.08174003660678864, 0.08110392838716507, 0.1227063238620758, 0.7144497632980347], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "51adbb7f2685badfd334acb97934b66e7bdb7e7ba2159309d4a2f8750a326566:action", "state_id": "8ff97a523284a5e6713c38b2b3a537672db9474a447b03660242ba85b8412627", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4609375, -1.48828125, -0.982421875, 0.607421875], "student_probs": [0.0869675949215889, 0.08462178707122803, 0.14033763110637665, 0.6880729794502258], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c91c7689012f458fe989eba59dbf9798b510c081f129784b300799d079689955:action", "state_id": "78f0c5065c443fad50cefbac4102400036f5d6261d78d0a4eb21b805cbfdba82", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.921875, -0.9453125, -0.24609375, 1.30859375], "student_probs": [0.07549089938402176, 0.07374215871095657, 0.14838248491287231, 0.7023844122886658], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e222e70bd851a9eba46995cc1acfde8f345fc57d969c7a0b9ae0fe1c7c397e60:action", "state_id": "c83674997974b670668ae9d9fc1883e4bf15aea0c160ce8a8ba7501761b28416", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.107421875, -1.07421875, -0.3037109375, 0.81640625], "student_probs": [0.08997097611427307, 0.09300844371318817, 0.20097851753234863, 0.6160420775413513], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "05310651c1f97a9a35501f90fe9420e7d4c29ee97a0fbed76df3a9e4f515d960:action", "state_id": "a64896b1366ee6eb7567ccd353dbbf4866210de6dbdf2d71b21b509bba426692", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.27734375, -1.27734375, -0.85546875, 0.7421875], "student_probs": [0.09041910618543625, 0.09041910618543625, 0.13787266612052917, 0.6812891364097595], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cec68a72e5d1366f1dbef8fe3a70699ee0b8ee48b1677c06c0868ba53cf7394b:action", "state_id": "bab5d298cd8e026388e3a765200dd818f08d6fd713c18b91f0e27527b07c56d3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55859375, -1.55078125, -1.2578125, 0.4375], "student_probs": [0.0932922288775444, 0.0940239280462265, 0.12602975964546204, 0.6866540312767029], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "992584d5a5b703fb7fef232e63d4746c73e94f1533cccc4174b7ae06c12330e7:action", "state_id": "17ee92c71ce4f2a6e7bc9a6628d72a78d28439e551cb23b10ecde4629da1066e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.62890625, -1.46875, -1.2734375, 0.271484375], "student_probs": [0.09719070047140121, 0.11407216638326645, 0.1386764943599701, 0.6500606536865234], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "da45a5f5ce5f1efd3548c6e9d4d05610f23575536d571f44e350f459f6329360:action", "state_id": "44f367a7a6b10b15c1b9ad7338f266d9d33f5495193dea6ac6de335593e76f58", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.59375, -1.46484375, -1.24609375, 0.548828125], "student_probs": [0.08281774073839188, 0.09421209245920181, 0.11724884808063507, 0.7057213187217712], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4ee4d5eaf6c7cbbada5c81f1c93868cff13b52151a07a1a7247848188bd2e445:action", "state_id": "6a64b24c5591bd54e47008ef4531d54579fcd4d7e9444c1ab0c9d78c0c141bff", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5, -1.3828125, -1.14453125, 0.611328125], "student_probs": [0.0846712589263916, 0.09519845247268677, 0.12081313133239746, 0.6993171572685242], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e898d63e44f8aa637461524dda243a1e8322e5236473ec65f36215411fb5a678:action", "state_id": "5d98be076c60d31f22defea53d5f858c8fcfdfbc3ed004a62d1dcc542e0d8539", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5390625, -1.453125, -1.15625, 0.4296875], "student_probs": [0.09330220520496368, 0.10167498141527176, 0.13681863248348236, 0.668204128742218], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "57b832e4f9f4247767c3683bff31c071ae0c23c9bc3499214afbb63a15d1634b:action", "state_id": "9e476843c4b838abd630acc1f134cc9e54130bc2afbef94f2c0cdad35ef7a285", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.56640625, -1.40625, -1.16015625, 0.255859375], "student_probs": [0.10141237825155258, 0.11902712285518646, 0.15223801136016846, 0.6273224949836731], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "346312faeeb8509c4fce2f08a5a0bded9d50db477ecb79a515390b23e8cceea8:action", "state_id": "da7fb1e078f12316c55a55a6998f97e05f7630c6ef21c9dcc89d8adbd472ed23", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58984375, -1.43359375, -1.11328125, 0.421875], "student_probs": [0.08884306252002716, 0.10386806726455688, 0.1430843025445938, 0.6642045974731445], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "70f0eb8107dec99f16fbf459f4cbdba5fde88623c9bf2960b5fb827829035e58:action", "state_id": "a0d948e37101f68635f12d08982661dbfe6df5a4c09eb1aa12092dee0f67e5a4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.16015625, -1.015625, -0.49365234375, 0.748046875], "student_probs": [0.09221791476011276, 0.10655760020017624, 0.1795867532491684, 0.621637761592865], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "31accba9b58f61544d965e5ff606a2b0677fca6875fc0f6e823aaa242632818c:action", "state_id": "dfed33f08ee732b28cbb533c92d95f4002992556ea4557312d9d842712f4aee6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.09375, -0.974609375, -0.505615234375, 0.8671875], "student_probs": [0.09063602238893509, 0.10210404545068741, 0.16320164501667023, 0.6440582871437073], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fdeed4c7e2c7b24f62d64ea078409e2c095bc0682c23a7f2c9c90d12b31cef2f:action", "state_id": "47c3df7728f2f2f47cb74ee33ce2d00a398f15e9e8db63efa88da7251d60ae24", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0859375, -0.9921875, -0.5584135055541992, 0.923828125], "student_probs": [0.08885267376899719, 0.09758558124303818, 0.15058138966560364, 0.662980318069458], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fd138b77e9a0c4bf817c46a1c45b2c4dab47a33aec8c675e3a86e437d4a4168b:action", "state_id": "e6e2f0bff6b0dfb0cdf4aebd2915669b3dfde89cf827f97179f990c5eb9e0e08", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4609375, -1.27734375, -1.044921875, 0.583984375], "student_probs": [0.08736682683229446, 0.10497365891933441, 0.1324402093887329, 0.6752192974090576], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "430e8913ee6810fff90cb2c7e6480cd88b61f807a0e6f510a250f12cb8f3a5e3:action", "state_id": "a580c4a7c542a0e4c4fc5b06aa93122b93000fad2dbce3eb84e2f63fae4a3d84", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.42578125, -1.27734375, -1.044921875, 0.87890625], "student_probs": [0.07328902184963226, 0.08501675724983215, 0.1072615459561348, 0.7344326972961426], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "411ba42154e6651e15112391fdfcb4af1e9b32afcf33738784b1fff43edf2be6:action", "state_id": "c2b074391b3b446116b874bb4d69c24feb74a926c1b9ba4d5a7469ad1764cf15", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.48046875, -1.375, -1.13671875, 0.681640625], "student_probs": [0.08189401030540466, 0.09100319445133209, 0.11548906564712524, 0.7116137146949768], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "641194c7f0fffc874cd46cfd3abec21eda2fe86073938e5cc49a03492b5d7bd1:action", "state_id": "bf61bb414ba5ae96abb84a76045da0325974ca5b5026990758ae469a8fe4f896", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.49609375, -1.46875, -1.15234375, 0.736328125], "student_probs": [0.0783676728606224, 0.08054010570049286, 0.11051613092422485, 0.7305760979652405], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1d3aad0a62a7d045633914474544a7e84dd327e3c3c32827c17396ff413d734c:action", "state_id": "5572c07e2208c999136d55f1e7ca1bcb239aaa8bbbf7904e22eedfa7abd510b3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4765625, -1.41796875, -1.12109375, 0.72265625], "student_probs": [0.07996699213981628, 0.08479255437850952, 0.1141008511185646, 0.7211396098136902], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "762450c7e45461ed6c2852402bb0fc38949169bd2dd636dcf0ce2a6d12979331:action", "state_id": "3884f765a4e2b3d820a65857b3f41c171536c38da50406676273b14896077f61", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.46875, -1.38671875, -1.1328125, 0.666015625], "student_probs": [0.08375345915555954, 0.09091351181268692, 0.11719216406345367, 0.7081409096717834], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "305b33bd08fd63900afae99790d45b8374b2e93b1a9e81df1b796e4319f2172b:action", "state_id": "05e7a64e84d7c82a85e61a0e461f4aca333bae484c017d230be87c418fb8a588", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4375, -1.2890625, -1.1015625, 0.708984375], "student_probs": [0.08254910260438919, 0.09575863182544708, 0.11550696194171906, 0.7061852812767029], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ee1c07c0f69b7a7bb5e7f5b69e07eb941c2d48234b6a177311675490b04e34ed:action", "state_id": "7deaf85a27c183d17cd7dfd4f3c9138e9a2e52d0174fea2cd9fd94aa52de5221", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55078125, -1.4609375, -1.27734375, 0.57421875], "student_probs": [0.08487972617149353, 0.0928587093949318, 0.11157229542732239, 0.7106892466545105], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1e2289a2e20ff492cc65e4cee6f66385895c7e40bd4be9773b828dfb68d53ebb:action", "state_id": "414befd3a03644753ab4098d39218108912dccb0efa25338fdbd55a4e1f37b6c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.64453125, 0.5234375, -1.62890625, 0.193359375], "student_probs": [0.05868706852197647, 0.5129550099372864, 0.059611253440380096, 0.36874669790267944], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f99d9b6d4ad4c73bb3ee89c388f1d76aa38451f2d11b5b3bfefcb7d6e7f39f6e:action", "state_id": "2ea4fed87e0559b95f5f5bfe82a3cc66f48c113085ccc8e7343f9a3251562235", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.40234375, 0.73046875, -1.4453125, 0.69140625], "student_probs": [0.05401960015296936, 0.45584800839424133, 0.051747605204582214, 0.4383847415447235], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "11b4c8dba17161e233707607b29f0f430021f131b237237fde67382e58a37335:action", "state_id": "7e4a28b341800de5cdeb278114a26d048684706f2b9c41f734aa4e8709995590", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.46484375, 0.2578125, -1.44140625, 0.349609375], "student_probs": [0.072670117020607, 0.40690773725509644, 0.07439343631267548, 0.4460287094116211], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "da1b8506d5342633083844be4a6e5ea9a8484c859e384ca231f84f526c301baf:action", "state_id": "020e4972eea847fcfc6f00ce7af5a2055c7b992bd8a97d510ad098c5d52b91a0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.36328125, 0.21484375, -1.265625, 0.32421875], "student_probs": [0.08094286173582077, 0.3922378420829773, 0.08924628049135208, 0.4375729560852051], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0e4d56b193dbac76636df0696ccb9cea9bf6c29c9bc08bf93eaf8d97fb9c37fe:action", "state_id": "3a98927f6ad4860ad41bcca5bbe70d7df3c56c7c6b1f5dff361213ddc9d5170c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.36328125, 0.25, -1.28515625, 0.396484375], "student_probs": [0.07744982093572617, 0.38874030113220215, 0.08374322205781937, 0.4500667154788971], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c147eb3c3a8a4b4347c575fef997d200a1fa8f791e9fdf7aa08b242c24d60793:action", "state_id": "de930f683429bad977f29b9c4b717832239aa278fe078b1d26bb80e46d4a71f9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.41796875, 0.203125, -1.32421875, 0.498046875], "student_probs": [0.07168078422546387, 0.36260586977005005, 0.07872594147920609, 0.486987441778183], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "26f6e8acf124351b4d93b0e9d396cdb472bd4db0c9da7bbbf1b3755f467b39ad:action", "state_id": "36c0651accd543717a16cabee0f823d1f2a4d14d839522a857612642ac6dad3e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.421875, 0.2421875, -1.32421875, 0.4140625], "student_probs": [0.07323693484067917, 0.3867437243461609, 0.08074984699487686, 0.45926955342292786], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b7a00226575b536a81637e2f36159ed58626df2f3a1eaee35184660aa7b6ba83:action", "state_id": "e517b55a7c5700bfb70e7d47e95994360239ae8821fba83212e501c9c3216408", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4375, 0.349609375, -1.39453125, 0.44921875], "student_probs": [0.06842868775129318, 0.40866735577583313, 0.07143306732177734, 0.45147088170051575], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e815c9c08500f0b54068c761d386da75ee467905d3f53976377e5dce95ccd9e5:action", "state_id": "85c7f14ad74867f678cb0dec1d8eae503b98d9328d0b70e290de613c090b2b95", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1171875, -1.42578125, -0.251953125, 0.529296875], "student_probs": [0.10754135251045227, 0.07898687571287155, 0.25547122955322266, 0.5580005049705505], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2416c4f3d7fc9355637efee2dfdc62ec7f0ed6acafd5229784348f24029e829e:action", "state_id": "eb03a4996d2587c8e128918e5e069239843ce0e6b23a5f3d8ca612f37279b3c4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.00390625, -1.087890625, -0.05859375, 0.8125], "student_probs": [0.09396038949489594, 0.08639147877693176, 0.2418181151151657, 0.5778299570083618], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3ea899d382e56227d6798f7ba58d4c12760861440d2bc002fa14a2a0dc3df30b:action", "state_id": "ab87e78b06c2590ae66656286f8d33cc4ab7374367a7aa5bc2a30c448514dfcf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0703125, -1.123046875, -0.09765625, 0.724609375], "student_probs": [0.09422764927148819, 0.08938735723495483, 0.24922843277454376, 0.5671566128730774], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cfe1c419479cec10cee20d679b5df808568f61eb25216a381d69064d92430daf:action", "state_id": "8f2845c3a66e44314d8ec1b8d697bcc3e4db3cf1c0934d1818829674079e6ab5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -1.03515625, -0.01171875, 0.763671875], "student_probs": [0.09319818764925003, 0.09229248017072678, 0.25682634115219116, 0.5576830506324768], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dc5a89e0e0a3585e9e2a1c061cf68e1adb3038338bcf466427662226f3300020:action", "state_id": "c4280f8326789bc3acbd4a418fe9457e5de1d47c826e9bc4ab4f9117fab7c31f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6640625, -1.75390625, -1.5703125, 0.470703125], "student_probs": [0.08720353990793228, 0.07971049100160599, 0.09577435255050659, 0.737311601638794], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "82078143802d070828712183dadd73e2099844ffb9e34f0f2e7092573417993b:action", "state_id": "7f15d0ad24146630bfa8a8a9d021b0cf5a0c72f8e1ede880fbf4e68d7e58d4b9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.59375, -1.671875, -1.3125, 0.44140625], "student_probs": [0.0917171984910965, 0.08482453227043152, 0.12150553613901138, 0.7019527554512024], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a304f09748a55ca4e7ef8f8f5c6134d6d3005ce9a313ff7304c7922eabed2c6e:action", "state_id": "75ffdd8f60b7a5ceee768e9ff18165d6d2ffb5fe9a647db09ab19751a6a27eaf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6328125, -1.640625, -1.35546875, 0.345703125], "student_probs": [0.09484212845563889, 0.09410405158996582, 0.12515556812286377, 0.6858982443809509], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "89ebba61d4410e84fefe831b9a441335f3f8eead1018bfe94e7847769c39ed28:action", "state_id": "3b8d69b6c0dcb256e7cfd9dfe50ab2f0657b708c9171f602eb39e9a56d97d4de", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.48046875, -1.4296875, -1.08984375, 0.4765625], "student_probs": [0.09426600486040115, 0.0991765707731247, 0.1393161118030548, 0.6672413349151611], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "984cdb32aaf2ab8cb7a877cf662200b29cda37d2d2c60c4ebb2e34d51ac834cb:action", "state_id": "21f7612ca4c3b82e9446d1550f1747821b88895e8def9711d1c120c0efe2f0ee", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.33984375, -1.59765625, -0.466064453125, 0.423828125], "student_probs": [0.09997492283582687, 0.07725463062524796, 0.23953479528427124, 0.5832356214523315], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cb771b9043624448a8b8494a91509a6329544bf866eba4eb99923da65152890e:action", "state_id": "9a2ea53079f05aca177afd5d3c0db47f320fa6c20237f5a4b8b66a9bf24d3a50", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1328125, -1.375, -0.2763671875, 0.705078125], "student_probs": [0.09594208747148514, 0.07530580461025238, 0.22592203319072723, 0.6028300523757935], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3d6c5fda23b5e95c7244b990995f807b91c685239efc0fdeceecad2002875025:action", "state_id": "ea08f46dceaf95a2ee6652fc4aa0c4d44bf228eaacab3fbbce8201ddacc4adcd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3671875, -1.34765625, -0.4833984375, 0.65625], "student_probs": [0.08330516517162323, 0.08494821190834045, 0.20160284638404846, 0.6301438212394714], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "94ebce94befbf2d54677db37fc06ca6bbf4659d60eca894a3a88112eae2187a0:action", "state_id": "f8b6cf0caf5bc47f669734e51e2de56da05a0a62ce56c871098694c7f5bdfb64", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1328125, -1.140625, -0.66455078125, 0.802734375], "student_probs": [0.09508175402879715, 0.09434181451797485, 0.15186603367328644, 0.6587103605270386], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e806266ae04272ba38d0330a26f11a17747ed4a30d553c4484b7b62cd6070097:action", "state_id": "9f30b4ad38403f636ae87448225a7a0fa7f71c256a6c59d23d822a5b925289ef", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3203125, -1.24609375, -0.9609375, 0.962890625], "student_probs": [0.07508904486894608, 0.08087407797574997, 0.10756009072065353, 0.7364768385887146], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "eaf7f6cdbe0cc1215c2a9029a77f1ea4f924107cc1b9786b5ed829392bf99aa8:action", "state_id": "a96d4defc16843e6cca154dde77e2ea37c6b91d82865c38280f3c15c93db895b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.34375, -1.26953125, -0.87109375, 1.0], "student_probs": [0.07091427594423294, 0.07637768983840942, 0.1137642189860344, 0.7389437556266785], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "797ad33f06856d0127b6faad517d5a06332ae49ba351f655c7bae86be6d583ef:action", "state_id": "d2d6b37ce87eebd9bcaf6a43c37d5514ac5a913bab37821563c12afef7311587", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2890625, -1.32421875, -0.865234375, 1.00390625], "student_probs": [0.07464051246643066, 0.07206201553344727, 0.11403568834066391, 0.7392618060112], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "482c840fb548fe815c3404871b5075d2b980804bd2408eb59c82ee3243d01f83:action", "state_id": "62cb2cb6a8ce6d70fd530678b1207bf1aa26f4a1d365940a4233f6ca7c5a7ec6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.12890625, -1.1484375, -0.7490234375, 1.130859375], "student_probs": [0.07678359746932983, 0.07529846578836441, 0.11226631700992584, 0.7356516122817993], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4dfee1441a791830f54684e0f26e11e9f8df8d49246ed78bee2cabb9e9cdc755:action", "state_id": "ce62266ad331580f20d0bd4ab722b5216c1039a2e488acb23316550b71f10d41", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.134765625, -1.16796875, -0.6611328125, 0.818359375], "student_probs": [0.09412787109613419, 0.09105384349822998, 0.15115216374397278, 0.6636661291122437], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "db71e09f4edff1b81627dd90173bc8c44ae67d5915348fd8a3902ba7b7621b4a:action", "state_id": "94c4ebc193cd81c577856724d665b37f0c3af1eace1b2a3f32d993674aa19646", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.07421875, -1.013671875, -0.596435546875, 0.939453125], "student_probs": [0.0895601138472557, 0.09515022486448288, 0.14441531896591187, 0.6708744168281555], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "782c021e033d9e7bc984d43c5f9ff29e6baf0a2f31232d596b632678d667c22c:action", "state_id": "a16c27c3d2fdada0f985a6d89dd4a1a1e5dc9c431a1686f84fe1e73fa30d25c5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.05859375, -0.998046875, -0.44873046875, 0.900390625], "student_probs": [0.09095276892185211, 0.09662981331348419, 0.16736945509910583, 0.6450480222702026], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "936c26ad028e36f41ad57e28378a547fae7245c10eef320478aae2297252a2bd:action", "state_id": "c3366767f21ce69f7909da9e57104a630ef3658e5bf32aa7db0c16a514db1936", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.11328125, -0.99609375, -0.34765625, 0.83203125], "student_probs": [0.08872731775045395, 0.0997588038444519, 0.1907937228679657, 0.6207201480865479], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "65ea654295d46c08de7a20af2dcb86dd219ff377cab946b6ac49ce40281ade83:action", "state_id": "32582a600b8594b201a215b8530dc1c63f6e3527fc3c0eba8a4aa8399976a6a0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.109375, -1.01171875, -0.4443359375, 0.921875], "student_probs": [0.08568421751260757, 0.09447403252124786, 0.16661866009235382, 0.6532230973243713], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "01df648f4a6e2fc56d5b8ad61a1c181de10af3bd9e2f2a3a03f80074f764bedf:action", "state_id": "279e1e03c090a71a97274be8437def747973bc5eace4c923b7f9485cfe04c36a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.07421875, -1.099609375, -0.673828125, 0.8984375], "student_probs": [0.093834288418293, 0.0914817675948143, 0.14003899693489075, 0.674644947052002], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "eba736dc362d1318183ce1bb7cf8026c6d1b8b6ed020a774e8ddaf20474dbe1f:action", "state_id": "f3854f7c53ba18a44f31850761dc15c659196fb3eaf0e60856fe0eb39c690517", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.12109375, -1.0546875, -0.5517730712890625, 0.83984375], "student_probs": [0.0913933739066124, 0.09766850620508194, 0.16149814426898956, 0.6494399905204773], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7d15549d9b978ccaba571e376a959c0f55f8e72a4d16c605b7f0559288501b44:action", "state_id": "922705297078b416f5f8dc3e7514aa4ca6330797ccec1197a68987cadfacf46a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2265625, -1.265625, -0.7880859375, 0.923828125], "student_probs": [0.08264294266700745, 0.07947694510221481, 0.12812496721744537, 0.7097551822662354], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "91cae6a5d1c016a5a7de2be016dc5b9c1be30fc7b1cd638aceb609a1d07f0f84:action", "state_id": "213629748de3938546ed559d9345f0660b3508d6272a92e352d94d72441731f6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.609375, -1.67578125, -1.19921875, 0.556640625], "student_probs": [0.08219452202320099, 0.07691358029842377, 0.12387152016162872, 0.7170203924179077], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6b37adfb3ac0658fe60c4302758ea1b075287c9153936ec76106ba6c7fbbb8a2:action", "state_id": "d1d1a6be8ce88bfe6adae3e58dc101e782a9aa71af85a7590d67dd73023c092b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.23828125, -1.125, -0.619140625, 0.984375], "student_probs": [0.0757053941488266, 0.08478602021932602, 0.1406099945306778, 0.6988986134529114], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "785f8d3e8b01c4e7477a1eef43359ab804b09a9722e0b6f7715dc6ec47bec83f:action", "state_id": "e626c3b9076718ea8ad2d258036094b2920c713555a2699819416c23607560fa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.171875, -1.2578125, -0.67041015625, 0.7578125], "student_probs": [0.09563751518726349, 0.08776192367076874, 0.1579107642173767, 0.6586897969245911], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ee1fc62ae8d40c094d6b3950e985904093d9b57916ba07a40e3209af87054b57:action", "state_id": "17964662d4d3bb25b5026c84f44c9c785fecc849c28f1fb3a1366dc7f12ba1e4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2890625, -1.16015625, -0.7353515625, 0.7578125], "student_probs": [0.08605152368545532, 0.09789079427719116, 0.14970357716083527, 0.6663541197776794], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b8ae3134dacf047a749dbc740b3c2d9924218f881e9be1a4ef44e8d19e991afe:action", "state_id": "dc7354597f3eef5134459da62f3113e03ace3242df3ea767055b67e34d92840c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.27734375, -1.16015625, -0.595947265625, 0.78125], "student_probs": [0.08378078788518906, 0.09419727325439453, 0.1656041294336319, 0.6564177870750427], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7d9491b68d8288102d17121227220266c3a5055b9a604005dc4af5d56060fa7c:action", "state_id": "bffa7a53f4827d3176ebba446db581c0d743a89d4c73342712a965554bb17175", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -0.96875, -0.3408203125, 0.88671875], "student_probs": [0.09170275181531906, 0.09799912571907043, 0.18362364172935486, 0.6266745328903198], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "596909f2cc04ad6d749c5b9bc8518715061b7fba3c6799a165e9b79010fb69e8:action", "state_id": "e6ba79686d62c24185b4f83c1ff044fc6352fc49dc24ca9c50d82f986a174166", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.138671875, -1.009765625, -0.5797119140625, 0.82421875], "student_probs": [0.09085693210363388, 0.10335735231637955, 0.15889540314674377, 0.6468903422355652], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8b9c184995c79fe6f3760427cc118ee03ccc693af1da5696909ecfe5b826f61a:action", "state_id": "73a3a68d69a82111a96c5ba3a7614253e1a896d0fdddd1863496f07414dea12c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1015625, -1.0078125, -0.5577011108398438, 0.857421875], "student_probs": [0.0916334018111229, 0.10063960403203964, 0.15785188972949982, 0.6498751044273376], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bfe966d2db4c293c3b965390bb1e784b589d321b1290fcc56c008d8ef1d75255:action", "state_id": "86f6119395ac5b7dbe4d4547b44a0ad653a2de5fd5b6a3b009c8a904d09bb7ae", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4140625, -1.2578125, -0.9140625, 0.634765625], "student_probs": [0.08638082444667816, 0.10098942369222641, 0.14241790771484375, 0.6702118515968323], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c009fb6e37cc176988d2bf644ec2fd2017d2237b2d7d222b794fee72e92b42ec:action", "state_id": "9ffefe4625804e8af2b253f73f594d54474f37c6f3306deb4ce1cda0d40acfcd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.375, -1.24609375, -0.888671875, 0.767578125], "student_probs": [0.0813981220126152, 0.09259714931249619, 0.13238050043582916, 0.6936241984367371], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e7a4a6261427bce9eb4e09b1461c2c0c093fe477b87bf7cd8d09e36b45392f0b:action", "state_id": "ffa4e0054be3ebb39c943511b1872508aa4cb4e1f5d2b883710468c2888ba5a6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3671875, -1.2578125, -0.931640625, 0.775390625], "student_probs": [0.08208315819501877, 0.09157037734985352, 0.1268848180770874, 0.6994616389274597], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "03c214e8eeaaa4bd22325174897f5def323d2afd8ce9c234711680934457eab1:action", "state_id": "838909492dc5ba6c6e86dd3f566e62f56760de72bd44c14ff2e2ae8b54a9ec42", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.34375, -1.234375, -0.8828125, 0.896484375], "student_probs": [0.07635526359081268, 0.08518044650554657, 0.12106583267450333, 0.7173984050750732], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c6f8ebaedadb4a38cf3e20621833d929c1c8d46e99471dcb1f0097119a6741e9:action", "state_id": "7fe43f06d7d87afcfcaa2372b93a14b8c3d7b3167fb274cb2d6913e352556576", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1953125, -1.125, -0.7939453125, 1.099609375], "student_probs": [0.0741269588470459, 0.0795266181230545, 0.11073572188615799, 0.7356107831001282], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "79e06620d18ac57676ca35a753da9e77e2ee15014787cfd086735b9ca47c9ce6:action", "state_id": "e1df2078880a7c438bde77fb9a6c4f47bb31418980bfeb79dc19d5c2e2156f5c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.22265625, -1.1484375, -0.810546875, 1.07421875], "student_probs": [0.07390926778316498, 0.079603411257267, 0.11160295456647873, 0.7348843812942505], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5d96ac9feb27d0765403a2e3be6ca1884933348d7bc7f38fcf26d5be839ce8fd:action", "state_id": "fea0493cbae0cee6c1f6206ec0063a4d296f7646e5335e4f6f0ceb45689a4f47", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.30078125, -1.23828125, -0.892578125, 1.029296875], "student_probs": [0.0722159743309021, 0.07687351107597351, 0.10862096399068832, 0.7422895431518555], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "32e199dbb0be2ef56abd15bc74304ad7e3e51f31df7874c534935c082898c084:action", "state_id": "3c9018d3e09eccfd426d46ffaea353fc31eb4ec43ac8c31f9f76d96837ab1aa8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3359375, -1.26953125, -0.923828125, 1.01171875], "student_probs": [0.07122667878866196, 0.07611715793609619, 0.10755226016044617, 0.7451039552688599], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1467f1f101b630d404059eaf79f2decbc9edcb9ad869a58ed3e6a1c6d25d6ec0:action", "state_id": "82ec46cdc583b205be6b57a56c6db22c0640fd2a69963c338b4dd06c60a967e9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37109375, -1.328125, -0.962890625, 0.96875], "student_probs": [0.07179971039295197, 0.07495209574699402, 0.10799485445022583, 0.745253324508667], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bd6660523fee368ae4dc614b0fab16507c0d2e32945f9da9c803a9fa1f91739e:action", "state_id": "fdd4c38ebbd29e622dafe91ac535c0d802215ac8496150db38127ec32d27869e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.359375, -1.30859375, -0.958984375, 0.962890625], "student_probs": [0.07276296615600586, 0.07655338197946548, 0.10859199613332748, 0.7420915961265564], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "58a6f3fbd9c063a0e016aab18d8abdb9a53db92f87788ab98f0fb13e9ef0241f:action", "state_id": "8dec472c1d446c8a48da575f4e36bd8b867e15f7c3007f75ddc789d932ed5bdd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4140625, -1.3359375, -1.046875, 0.853515625], "student_probs": [0.07586727291345596, 0.08203208446502686, 0.10952720046043396, 0.732573390007019], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ed73c88f9a2d07ba0097aafbab90e29e3ed45cc3e6756ac36156075476b81d45:action", "state_id": "fdaf26502d0d261776b24adf99265dae236121c7bfaff8a63f6e6410506e787c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.421875, -1.35546875, -1.041015625, 0.876953125], "student_probs": [0.07410315424203873, 0.07919113337993622, 0.10845305770635605, 0.7382526993751526], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "64f1b4b90713a76e30eca95f23622dd2f42c0f541c9be7b10396abb78e591b08:action", "state_id": "d9b1707a4bfc25d8ff71c98e0117922570c1f653b231a6027bc944c54b507a52", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.43359375, -1.37109375, -1.0390625, 0.833984375], "student_probs": [0.07573365420103073, 0.08061805367469788, 0.11236514896154404, 0.7312831282615662], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c1eff01e50a6696d5d7d2a5cb54e0783ca1a9bdd915f473fe9ea7380f3421848:action", "state_id": "99485e1c83d80d17847c03666b3a8d0da6d0e55fb02c046c471e7a549fbb3fa4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.46484375, -1.37109375, -1.1328125, 0.736328125], "student_probs": [0.07982325553894043, 0.08766869455575943, 0.11125736683607101, 0.7212507128715515], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "52fb259d164282a45c2af22416ad30adf265336cf181001cd0c69c6f3e738273:action", "state_id": "3b498bca2cc4fa79202bca1e8e77cffd4f6b1e9bec013a82a9016a23fe48e46a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.44140625, -1.25, -1.0703125, 0.845703125], "student_probs": [0.07403730601072311, 0.08965557813644409, 0.1073036789894104, 0.7290034890174866], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "50a765d206325f02be5945c924bcb6eb638942ee368d373d52f5b85b40697fcc:action", "state_id": "629b69ff8bb5125580f6ee8190a4bb677aaf7f851b82793e1623832ea66e77a0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1796875, -1.48828125, -0.4306640625, 0.435546875], "student_probs": [0.1126319020986557, 0.08272577077150345, 0.23820899426937103, 0.566433310508728], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7249164131f213e91e0de028877ad71ea938ecfd9fd31f3e7b5066d6c95ac659:action", "state_id": "e3a9c724892eeb7d2f21e12d3b5ba33ede4664fd128a976bc9055d44acf9e662", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -1.138671875, -0.185546875, 0.720703125], "student_probs": [0.10147976130247116, 0.0897306576371193, 0.23274362087249756, 0.5760459899902344], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "810fdfdbf72e8ce926bfbd202bff31f3a7a454859e1fab4997846b38870f532e:action", "state_id": "a91145d6bccc82907e475bfa56f4b8ecf604454d7c6a7d232e740bf2bf89d6a4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.076171875, -1.17578125, -0.130859375, 0.69140625], "student_probs": [0.0967542752623558, 0.0875810906291008, 0.24900849163532257, 0.5666561126708984], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "20d64c6e18fbc3d3628ac312435ca008f1fdb9d98b904a70c729433cc0035a96:action", "state_id": "c663e78b3f2e1b5d8c8d20b577e2c85f80b18d10a98d83b157c0bf8ab3893a79", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -1.04296875, -0.041015625, 0.755859375], "student_probs": [0.09521330893039703, 0.09264509379863739, 0.2523278295993805, 0.5598137378692627], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "07bd0c7ac4e85caebb4cc2c4fe3651a434ebb32cf29b3a01f945e3e8e6436915:action", "state_id": "59cbde6142123db56e822e4b5b43dc70cc11a8cb122ecc5fd4059447513cde95", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.083984375, -1.4140625, -0.220703125, 0.517578125], "student_probs": [0.11049024015665054, 0.07942785322666168, 0.261964350938797, 0.5481175184249878], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "84704ba628510c0f5b3cf3fb81a04280a0f89da76c9084f072afd4a699cbf473:action", "state_id": "1891249d1917780b7a6ace8adeefb2d1696989e4dc966bb7dc5dc02d7db9f49a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9609375, -1.056640625, -0.0390625, 0.796875], "student_probs": [0.09782370924949646, 0.08889570832252502, 0.2459287941455841, 0.567351758480072], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "089f521ec6726f712441e33239a1f25f9f7c0f4766ffef0d57303ef79cc7b53b:action", "state_id": "f8a7f567130e218251d1c643a026c1ca5c4c3a05dbd51a0a637d6c162e47a247", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.05078125, -1.04296875, -0.064453125, 0.73828125], "student_probs": [0.09369388967752457, 0.09442874789237976, 0.25122806429862976, 0.5606492757797241], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fe0bbd8a5a872fdb488011a7c25245dab0f39dbb3c2231ab0f210becbe608506:action", "state_id": "6a71b0d5b9cb74e987af9aa57da93fab677489c70feb80b80173f18f1e42cdf7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -0.94921875, 0.001953125, 0.798828125], "student_probs": [0.09000777453184128, 0.09751188009977341, 0.25243306159973145, 0.5600472092628479], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "84f4ff9f8b0046dc9b762ad91a1f4fd4db87d38b199c6200054c4a5fc54ffc15:action", "state_id": "2ea65520e867513a6c9b21b7205568ffed797c4efae1e15a0bf26404599f086c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.005859375, -0.919921875, 0.01171875, 0.841796875], "student_probs": [0.0892765074968338, 0.09728801995515823, 0.24698224663734436, 0.5664532780647278], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "73411554a84ad901c682213c6de89f1ef3cc459bd49f3d141e84750fb36218c9:action", "state_id": "3eed9b26941f5e5db66cbc60aacb14b99ecf9fa327f29f2458c4eff3bb8d2104", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.013671875, -0.923828125, 0.029296875, 0.85546875], "student_probs": [0.08760897815227509, 0.09584452211856842, 0.2486017644405365, 0.5679447650909424], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a31a93f6d1fa79e487e7258527a01a5b7d3735bb983d562d044102b404c136dc:action", "state_id": "fe262b8c9f9cdb00cc120c0f6548f7734dedc9b14a58230f7bb9936a972d3c0b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -0.884765625, 0.05078125, 0.82421875], "student_probs": [0.08818121254444122, 0.10050962120294571, 0.2561594843864441, 0.5551496744155884], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "20abf59e1053c30d2410336cd14fc4b88c797b80006eb7c69bccd7fd3c05a878:action", "state_id": "22f816f5fd36464dc704a8e44a0b5d88d253f94aebc95365283bac56db090db3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03125, -0.8193359375, 0.0703125, 0.841796875], "student_probs": [0.08508465439081192, 0.1051681786775589, 0.256008118391037, 0.553739070892334], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ad8d8b4d88f49ab0573eacea05df488688ae09ab704ba69e2c687562458ddc86:action", "state_id": "70e04afabba1a7e702d07af193bb9f79f69091a58b03c46504723bdd84d17788", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1015625, -1.40234375, -0.205078125, 0.578125], "student_probs": [0.10465624928474426, 0.07747070491313934, 0.2565094530582428, 0.5613635182380676], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dda77f1f1f9ae2f075eb6910599a4312360be35eb5b4f2bb1e73332de52bbc3c:action", "state_id": "77b7540ba7fd081e4399f932047ce7ad00d7e5daecc391264cc7c82009cac2fe", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.986328125, -1.029296875, -0.044921875, 0.853515625], "student_probs": [0.09244637936353683, 0.08855821937322617, 0.23699407279491425, 0.582001268863678], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c299b9e03dd591c4d6b4897b36152ce768a911855c27fdb6b615d20a87345a6b:action", "state_id": "5dcc385c550124dab394a3dbe6bb6481068e5be8f3ac2a4cc1f725ef6f328f13", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0546875, -1.03515625, -0.046875, 0.77734375], "student_probs": [0.09086047112941742, 0.09265252947807312, 0.2489214837551117, 0.5675655603408813], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c637cc7e6620d5ebaebc06f156fa3a4b2c7d5ad3644a8aeedc39cc00109549dc:action", "state_id": "6f42585b78beec13b9da240323d57c899585f414186f87715e01e49c012480e9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -0.94921875, 0.0078125, 0.822265625], "student_probs": [0.08869819343090057, 0.09609311074018478, 0.2502220869064331, 0.5649865865707397], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "78a9dd646d359b74635267bda7b8f27b9a9fc42b4afa99f4ce51870f320d3ad6:action", "state_id": "3284eb0584dec1845dbd79404d0d56f686bf837b015442611846e8455a8049f5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.005859375, -0.931640625, 0.033203125, 0.80859375], "student_probs": [0.09056882560253143, 0.09754645824432373, 0.25599873065948486, 0.5558859705924988], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f60506fc3ae824ff4cac81e9337e75c86d8ace0902c6d98beac9f9b7b40d75e6:action", "state_id": "be50636292919f399ee9b8daf857e8c645cc6b06a2d99f70759c3dfecdf422e7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.001953125, -0.923828125, 0.029296875, 0.81640625], "student_probs": [0.09051765501499176, 0.09787292033433914, 0.25386303663253784, 0.5577463507652283], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2c19980d818903132d1757705a5970add3306e84811364df889c06d3fa2bc583:action", "state_id": "e9cdd8d48d8998055fbedf47f9c047b92d0da5eb01520cb058e58c900be8b640", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.984375, -0.91796875, 0.01953125, 0.81640625], "student_probs": [0.0921492800116539, 0.09847631305456161, 0.25146809220314026, 0.5579063296318054], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fd20b431dda9854c1e06a39cc2678635193b56855b569c88ebdf90aa404397a2:action", "state_id": "76e79b74d87b56c025748082e1438eaaf4dcf74b7f1720753cef77270cfa405e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0078125, -0.900390625, -0.134765625, 0.865234375], "student_probs": [0.09077957272529602, 0.10107433050870895, 0.21734397113323212, 0.5908021330833435], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "13a958319394663e507c84b697d5edc4b014390ab99bc881241f609e5e984579:action", "state_id": "1760a87cfa2e23afaad1abed67b5c4d7f1a6b2a4b585fc93d622ffd7ca460b35", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.009765625, -0.87890625, -0.20703125, 0.890625], "student_probs": [0.0904163345694542, 0.10305722057819366, 0.20177623629570007, 0.6047502756118774], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9f3cd630c7c5bd883ddd9cfbacbb122a41df0c81e9ccbf646c7a0bf1a8c092e4:action", "state_id": "6719d0018c9c9bf9fdff430476ee1bd184acd164de49a39afc347dec225def55", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.98828125, -0.896484375, -0.24609375, 0.875], "student_probs": [0.09397156536579132, 0.10300619900226593, 0.1973896622657776, 0.6056326031684875], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3c2bf86605944f617feffdd76d9104dfc7d033cb6a27a0452b3380c5732dd173:action", "state_id": "108e4ec9d3b737755654b4b7ae63d82e97c754bf6f61c0321c8aecb1b46fe6bc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.02734375, -0.9140625, -0.3046875, 0.873046875], "student_probs": [0.09201027452945709, 0.10304661840200424, 0.18953174352645874, 0.6154113411903381], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c9bcbb73408ff696b301fb403719cad99b981ab91e34e4d72e1ebdb3c76de79a:action", "state_id": "d68b562f507eb0e011d2a5b42a071eb26e93a66e39c58d200b472be229d4841d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -0.94921875, -0.3330078125, 0.83203125], "student_probs": [0.094202920794487, 0.10305831581354141, 0.19085346162319183, 0.6118853688240051], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "10853ad800d74f6ce5d7e7350f6d964000f3b01647c00b3397de1f270ec51147:action", "state_id": "513e04f1a83a91fb76a9d805f955e0cd76bff514b9c575d164786bdb75101e96", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.064453125, -0.9765625, -0.3046875, 0.771484375], "student_probs": [0.09523216634988785, 0.10398101806640625, 0.2035849541425705, 0.5972018837928772], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "50c75ed484a24b4cf7ca48fe24a3300347150200df5f9e27faa70d69257cbd96:action", "state_id": "8acbce309bf839934223dbeb31d43338011e8096e5cb397ee356a568fd249aec", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.08984375, -0.99609375, -0.23046875, 0.720703125], "student_probs": [0.0945737436413765, 0.10386893153190613, 0.2233532965183258, 0.5782039761543274], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4b8364939c6e90146a4b05c020b8247922c6fae25517c21a6fbb935eb96e5935:action", "state_id": "493c03c5a8953a568e2a7bdc52514a5d544dfd8205be6d30ecee2ca23c2ca412", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.056640625, -0.994140625, -0.134765625, 0.720703125], "student_probs": [0.09530508518218994, 0.10145172476768494, 0.23959693312644958, 0.5636462569236755], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "93d4a52ff1ed08cbbd4c3f5869ec3e5c4356ab02b380cfb131dca416432142e6:action", "state_id": "f655e6a490c9405353ab1eeb230b47a6dfcafc317ba811c66c2ba751052cb8ed", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.02734375, -0.96875, 0.0, 0.7578125], "student_probs": [0.09246852993965149, 0.09804848581552505, 0.2583233416080475, 0.551159679889679], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3eda36baa1d48cdc1b4bf669de463ed9084ce185a10aada20cf0018a5c368652:action", "state_id": "0d4830c0d0d8c620b935fc9d9a1c403066ec18fd047aae0b9f4903e74bc3981e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.96484375, 0.009765625, 0.740234375], "student_probs": [0.09259733557701111, 0.09914860129356384, 0.2627568542957306, 0.5454972982406616], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d6c3c38a424a974fbeccd73a749f8d50117eaef0f63ba9ede5d1f0e36143827e:action", "state_id": "26789e44f99a70f38db33f71a3811888b204b93f7aacf90c6b8e9d2c8966284e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -0.955078125, 0.01171875, 0.7734375], "student_probs": [0.09111330658197403, 0.09813288599252701, 0.25804123282432556, 0.552712619304657], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "634d324ab0b2ae479f1c68df0a9c12dc1aef03f0fb010db0ddbd7cc39a3a4e1b:action", "state_id": "88e0374866f851f950cda20b71ac12b846aeb8ac1f6d52738b842c7c8c65695c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -0.947265625, 0.013671875, 0.7734375], "student_probs": [0.0900326520204544, 0.09888152778148651, 0.25849074125289917, 0.5525950789451599], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a45ae6dda8f0b6932b16ec6e55e55265eb7a05a8f9ebdd3e2b0cd9aa30817822:action", "state_id": "81e56c7a03cc61e496a5b5ab6070ca96d8b3186cc568595b29d117537ad75530", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.037109375, -0.9453125, 0.021484375, 0.7890625], "student_probs": [0.08937729150056839, 0.09797021001577377, 0.25761348009109497, 0.5550390481948853], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1e9cdfcf7245c1bcb0376c5f16ea0b19c42a687cc4385797e52c1bf3599a4354:action", "state_id": "f9c640cd3923b60fbcb10be7c37ece2098d5d2517753eaedcf7f511d297569c6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -0.94921875, 0.029296875, 0.8046875], "student_probs": [0.08798783272504807, 0.09663572162389755, 0.2570997476577759, 0.5582767128944397], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3c440035ea9114f4caf35b7ecfb70b7c496668f079c2ed1fb6765a382d1a09fe:action", "state_id": "55d70600c93aec30a7b87ef31140bf06d4cd0b4ff8f103baab246b1335f432a2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.060546875, -0.931640625, -0.005859375, 0.78125], "student_probs": [0.08836635947227478, 0.10052411258220673, 0.253706693649292, 0.5574028491973877], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6a31294df63e937ceaf56536e809b53a894e3015769c04b7d7b5e8994dfae314:action", "state_id": "b2fcbf1d22fd5e3429ca85b7ee648cfc867ae4faf81a7857320521806569fd0d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0625, -0.966796875, 0.001953125, 0.794921875], "student_probs": [0.08766636997461319, 0.096470907330513, 0.25416699051856995, 0.5616956949234009], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2722a5ca5445d04cc30aca0a1d4f4b14b9884dd199b5f31013d36fd793b8e762:action", "state_id": "5aa83505894b8da9deebd25db893fab5157086415c9b410fea6a2033736588e0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -0.953125, 0.017578125, 0.8046875], "student_probs": [0.08828553557395935, 0.09658467024564743, 0.2549642026424408, 0.5601656436920166], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b84cfb2c63a0de368ca4560f4fe90afd6598b91743d1b7e4847be1f730daa1e5:action", "state_id": "983d96af858cb19f801fedee5c945fbf06b3bc9ec91043b8c685bc3130cb9af6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0546875, -0.98828125, 0.03515625, 0.794921875], "student_probs": [0.08772078901529312, 0.09374377131462097, 0.26086491346359253, 0.557670533657074], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0535adc9decd3f6e3091b0f50b337e9ab8626a03b7e14839e551fc048357078:action", "state_id": "e500b190fe1f80764c66608ee25b9fb82cba758646838604f15eb74ead8e0191", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.96484375, 0.05078125, 0.8203125], "student_probs": [0.08827374875545502, 0.09378356486558914, 0.25894472002983093, 0.5589979887008667], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "77d830b325f57ca4e6ad8f6084386dfa1442b8ecaadf2b5b788e3ad0968323ee:action", "state_id": "2e0d6bb192684690f2500d3ad4af4b994bcc408c1087b0a997c8c4efcd201dc6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.037109375, -0.982421875, 0.03515625, 0.8125], "student_probs": [0.08821813017129898, 0.09317691624164581, 0.25777268409729004, 0.5608322620391846], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8843562525485d639cb6f1bac8e81aaa3fb13e5fc84c3f985807cf1e10520e34:action", "state_id": "6847c641758865a911cb8667352def97c579bd4cd25874294aaad8a4c94d0868", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.958984375, 0.00390625, 0.8125], "student_probs": [0.0896778330206871, 0.09583517909049988, 0.2510169446468353, 0.5634700655937195], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4190435f7739261cc2386b6d7c75e70ef7652f38ce28674264cb9f6d3bbf3b16:action", "state_id": "6ea586287aa82134ac85cb0f8b0db59df3a6ec7af7f2f169076620e7ee3579a3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01953125, -0.9609375, 0.001953125, 0.822265625], "student_probs": [0.08972214162349701, 0.0951363667845726, 0.24918657541275024, 0.5659549236297607], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "db9acf3acf89854ca67456c43bd6d374e736dbaa605a271002bc58eefd4ecf0e:action", "state_id": "35b0af410232806a64a8aeea154f582ea31268c62450c0c2b178ac1ea518a470", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9921875, -0.966796875, 0.041015625, 0.822265625], "student_probs": [0.09112874418497086, 0.0934721827507019, 0.25607654452323914, 0.5593225955963135], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b22f4b37c9635845b092dc4031fd578664cd67f6bb52ab654a32e2dc7c160090:action", "state_id": "fc8a885d66abf58b2756aa11c05f67d46d60e88c9cb00ac653e60192ce4e0762", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.98046875, 0.015625, 0.802734375], "student_probs": [0.08944972604513168, 0.09429338574409485, 0.25531673431396484, 0.560940146446228], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "df55133725ccd627db060b2f8a66f4095ad930ce11df52b3d2a2864b2db0e7c2:action", "state_id": "9427030a1b02bb887b1b68eef124b5613361c4c2a84f79ead87df56b9cf052f8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -1.013671875, 0.001953125, 0.77734375], "student_probs": [0.09133204817771912, 0.09313341230154037, 0.2571495771408081, 0.5583849549293518], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9fae8d0c50f4d6f375f4e8c10755129c2f298ba4cff8daa29f30feabf1e10d5c:action", "state_id": "27ab8b9800ed236b615cfe77463b729dbbaab2bce7078f6fa06c862a6a4d5e1a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01171875, -0.953125, 0.025390625, 0.830078125], "student_probs": [0.0893714427947998, 0.09476450085639954, 0.25212135910987854, 0.5637427568435669], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f99a2322845fcfe092801303d450c814f9f6f0163dff33a7bc8d2d29a1c35b17:action", "state_id": "4a0a31ab9bc08601d6a0bf0ba2af06c022173ca28988cd037be25c156b5c7736", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0078125, -0.93359375, 0.041015625, 0.845703125], "student_probs": [0.08838947117328644, 0.09519920498132706, 0.25229042768478394, 0.5641208291053772], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e85994e9fa726a2ac9a43f805c368acf1861b8b128fd75b23d4a4fcf6024aa54:action", "state_id": "dbc5e78fdf0c2d3fa05759930cf0372e4910ce97ad79b20761b6baddca63c3da", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.955078125, 0.015625, 0.822265625], "student_probs": [0.0888899490237236, 0.0953649953007698, 0.25174450874328613, 0.5640006065368652], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b49932160536b35a8e0ce8331ff7d8759f8649ac559f2cc2f935aa94fc9afccb:action", "state_id": "b426af89eb006f0c27775729d0cdaab0cd48af9c853f6d8e137292c8419338d8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -0.939453125, 0.033203125, 0.837890625], "student_probs": [0.08837302774190903, 0.09536758065223694, 0.2522434890270233, 0.5640158653259277], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c99741db55e81b07b5a49f566a5ec7e9ab058d32398ff93d5a92c7ce21715552:action", "state_id": "569f8776712ac18170da2de90bb9e3a3ffa54ac7c175ebfb7e5d6c987ff844ce", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01953125, -0.9296875, 0.015625, 0.806640625], "student_probs": [0.08993114531040192, 0.09838496893644333, 0.2532052993774414, 0.5584785342216492], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81c5ac54b0037e66d64a4a88b0826e95168dca6fbb29a1fb39ae8b9d29ee5f02:action", "state_id": "f428ffa9fe0f6d429197d51a7b976e1ed8c45ba41b8f9136e66cc8b2ebe88578", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -0.953125, 0.037109375, 0.830078125], "student_probs": [0.08879007399082184, 0.09451653808355331, 0.2544257938861847, 0.5622676610946655], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5dc4aaf121b9c3dcc983c23f973819a712f95b81e0627d388d4498166cb4d487:action", "state_id": "6e12862d243906cab61fc69e6aa6f228d3c0eb94b184cf4ff0eff8ebb0330321", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.021484375, -0.7890625, 0.02734375, 0.826171875], "student_probs": [0.08725327998399734, 0.11008325964212418, 0.2490474134683609, 0.5536160469055176], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c818a52e8700df49b93fc4a10f269bd8ab66305b56b5f3a51a1546991d275b14:action", "state_id": "8febc4b1db4f3f7cddcb09c414b7185d68360a6a0d83382d6d8e7e3cc69ad9ab", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.974609375, -1.32421875, -0.208984375, 1.265625], "student_probs": [0.07546694576740265, 0.05320143699645996, 0.16227944195270538, 0.7090522050857544], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a2fb7217256d4a08ac6d2c29d21280a365a37cc6185eaaa28adde3a41a4305c6:action", "state_id": "f4c1c9e08b2ac7d882928663ad58be80713af8f6e0b127bc62468f1bd99f79af", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8828125, -1.060546875, -0.072265625, 1.30078125], "student_probs": [0.07713396847248077, 0.06457383930683136, 0.1734849065542221, 0.684807300567627], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "98cd7e730c2d6f553c44d2c7c7ec3a23ec966dcc28763af8242c717c24839405:action", "state_id": "86319e90a5f15e1c583a15bf7156b917856f13213b014144d7e4f06ff3e7407f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.958984375, -1.1015625, -0.154296875, 1.287109375], "student_probs": [0.07377969473600388, 0.06397583335638046, 0.1649712175130844, 0.6972731947898865], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4a76544217693d5038b2e4124738eb0d0fca6ef07de0fab5566c19715149f3ba:action", "state_id": "a6d72f6cecea4e239fb6dc76257232ff62aa990e7ee4baf52d2bebd79554e37f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.017578125, -1.083984375, -0.208984375, 1.1484375], "student_probs": [0.07749560475349426, 0.07251656800508499, 0.1739581972360611, 0.676029622554779], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "54b6be20bd9f5e57e1191a59d40d532ca28742556347a83066e7f27677a3c042:action", "state_id": "b05d07686bafcf5a298be7c1e75191f7a8ff5de65a401924e45e693bd36661c0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0546875, -1.07421875, -0.24609375, 1.15234375], "student_probs": [0.07510833442211151, 0.07365560531616211, 0.16859936714172363, 0.6826366782188416], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1793afffc73e09c0a3da96f4207cc6346eb7f678e5415d36aa78506eee91245e:action", "state_id": "f1c69933b36d0c870f9ef8b71f17ee7e0fb978a100273cc4fb0f305d80c17237", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.15234375, -1.26953125, -0.3896484375, 1.173828125], "student_probs": [0.07006659358739853, 0.06231851875782013, 0.15022608637809753, 0.7173888087272644], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6324af1398ef854be15f7a69a2bce416cb1930db8d0c5619bc0ef6f034ee66cd:action", "state_id": "d897612efd33a33e94d44798b532a6a25dea55a9bfc85516657efe299b8ace26", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37890625, -1.5234375, -0.865234375, 0.751953125], "student_probs": [0.08361940085887909, 0.07236656546592712, 0.13976290822029114, 0.7042511105537415], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81ae09864bd89325355c3853843919c31aeb683a1e033d55e9cbe24f9858f0ee:action", "state_id": "54a86be453dbab7bbd39cde541286a76aa161056fe1a317dc2989089e541a36e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -1.115234375, -0.37890625, 1.044921875], "student_probs": [0.08389624953269958, 0.07789503037929535, 0.16266457736492157, 0.6755440831184387], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9a95dbbc9d5c1c86ff386664a87d79b0a68aaabbbabebc96dc9c2fb107997f51:action", "state_id": "671b9b8b1d24d2f8b0d8ea9313fa8195ee2bf84036a546dd8620374c4287942d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.29296875, -1.36328125, -0.76953125, 0.60546875], "student_probs": [0.09713096171617508, 0.09053601324558258, 0.16393953561782837, 0.648393452167511], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2c333cde656b52f786c53aa28028ba937d5d828fc8fa7e01feabf1717a1ded59:action", "state_id": "0f11782cc354cf64fb3a4564a1c8ff67a37adbdccecfba174b4c5f4c0db3cee8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.56640625, -1.3828125, -1.052734375, 0.4765625], "student_probs": [0.08630794286727905, 0.10370137542486191, 0.1442565768957138, 0.6657341122627258], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cc4cbababe34a2b1bc4e59c614634b47c349a25176473688d84fc6eebbae224e:action", "state_id": "fc2ad8488699497e1f1560160b51648ab264f5afbcec9a2e3a5ec2e2321c4be2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.15234375, -0.984375, -0.474609375, 0.7890625], "student_probs": [0.08992115408182144, 0.10636769235134125, 0.17709168791770935, 0.6266194581985474], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dc1758e1e44814ed9c4c93c4c615fe0d39266cbdd4f9f9e11dd6f853f6c30cea:action", "state_id": "027ae1d1b121079235da586ff54bf32b3a94bd89149241552611c41bc56a1350", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.052734375, -0.966796875, -0.3173828125, 0.876953125], "student_probs": [0.09038884937763214, 0.09850018471479416, 0.18857060372829437, 0.6225403547286987], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "75453e1d010365f17e02a9ba7c30ef06e60fc8852e4be087770a50531ec85b86:action", "state_id": "fa8e20eba61f2c1e95e3c9c3362fca35b902f644f5d5b721d11e0a5c574fb34d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.083984375, -0.9921875, -0.47119140625, 0.859375], "student_probs": [0.09154248982667923, 0.10034358501434326, 0.16894887387752533, 0.6391649842262268], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4dab39de8fffbfebb1910b34a29e5854bdee67e8ca1972be2565df26802e6429:action", "state_id": "90b8066b6888427776ffed2f75ef65818b7e694febf8f005eb838f2759ed0fc4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.296875, -1.16796875, -0.8115234375, 0.7265625], "student_probs": [0.08828765898942947, 0.10043457895517349, 0.14344502985477448, 0.6678327918052673], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0e9b16afb6cb531a6c052076335c3f0a74fa6dd7c87f95d90a7b803e170dd2ab:action", "state_id": "2a89518886de93d504534d1f90385153dd0f8fc1f0a5460fc051ba6e5806bfed", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37890625, -1.2734375, -0.921875, 0.587890625], "student_probs": [0.09226493537425995, 0.1025276929140091, 0.1457212269306183, 0.6594861745834351], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "34d126f66d2f5061be3521abdd8cbb6438c66012f5adedd757fce014391e3130:action", "state_id": "5ed4dec8526230b2cdb5458fea9da7535455528894707d933e07a5f187854d5c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.359375, -1.265625, -0.904296875, 0.728515625], "student_probs": [0.08516103029251099, 0.09353108704090118, 0.13423903286457062, 0.6870688796043396], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bf097aba88b8a94743d24f82743a2ee20f0714c4befaab41a147ce06ac5c1b93:action", "state_id": "ba8599eead652695fa27d7c5a6d0c36f68f0f3cd9e7d86b945c250d77d1c9555", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2578125, -1.21875, -0.8203125, 0.91796875], "student_probs": [0.08065991848707199, 0.08387304842472076, 0.12492852658033371, 0.7105384469032288], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e1ed65fcf46f023f2f7cb1469124e4440ebdf08ed37bdc4fd95e7de26df4dea1:action", "state_id": "aee2b64dccfb916e297804243c1e4157d46f2bf39debb9e26c45a9115a9647ae", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.123046875, -1.091796875, -0.69384765625, 1.0625], "student_probs": [0.08023568242788315, 0.08278264105319977, 0.12324418127536774, 0.7137374877929688], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cbd98238bf19cec18d31b28299cc90ee474aa744f804172303061109efc7825d:action", "state_id": "3a6d557ad5e8b016067faadd2cfc0f490bc0d3b6f19ede864badb94051e91478", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3125, -1.22265625, -0.90625, 1.015625], "student_probs": [0.07218199968338013, 0.07896734774112701, 0.10835801810026169, 0.7404926419258118], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "623c7c6ba0378f8c24344f7dd640f9d5abe22284e5b729d2d0a09c5e415a1f3d:action", "state_id": "3b9456f562bddf8fdb79b69611e4d5747491ac882521915a4f0e6d8c8871028b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.265625, -1.1953125, -0.83984375, 1.115234375], "student_probs": [0.06935860961675644, 0.07441092282533646, 0.10617318004369736, 0.7500573396682739], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ba0add295d1d10fc537217f8e0ecb21a46dd445b5483782e98956b3d14216bc6:action", "state_id": "a8b6f4514eaaa7eeaf865c039c621624507e0c791a92c22cc5ab316eb1875053", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3125, -1.21875, -0.904296875, 1.095703125], "student_probs": [0.06795153766870499, 0.07463016360998154, 0.1022067666053772, 0.7552115321159363], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "93169ebf75c94f30bd3a2c01f5678bd49e32f4b41f9679747f0d375edffb7ba8:action", "state_id": "894458ba8116e675f7c672b3b64a700e6c58a53237048fa5cddc7fb45bc60e5f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3515625, -1.26953125, -0.974609375, 1.037109375], "student_probs": [0.06924088299274445, 0.07516027241945267, 0.10094185173511505, 0.7546570301055908], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9ee9620a0067be315d4e302d15f9feea69f5a70895769e492d7b7f38ecb7db9c:action", "state_id": "f0ca10ae54b8849f50a6c9bacec87760370a65a9164df4cd0774dab3d12467f6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2890625, -1.2421875, -0.89453125, 1.0390625], "student_probs": [0.0725143700838089, 0.07599440962076187, 0.1075887531042099, 0.7439023852348328], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "012b4cdcfc187ebd92f1a23df246a166c9b178082a8960bd5725d3ec67d73d12:action", "state_id": "300cb157669a1af79bff23fbeeb93f138414670c93f1d59b4f16ca259009ba56", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37890625, -1.31640625, -1.001953125, 0.921875], "student_probs": [0.07405044138431549, 0.07882627844810486, 0.10795339196920395, 0.7391698360443115], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5f8825ba198e4824bb9ba5a81f6d87f8366d63307cc1728b13209fb6d4b0bea3:action", "state_id": "21da7591a54c9692268d75a86460e48fa287b99a04b2123aeb411f7558040557", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.484375, -1.44921875, -1.12890625, 0.76953125], "student_probs": [0.07699710875749588, 0.07975218445062637, 0.1098632737994194, 0.7333874106407166], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ef5eac49fd7280dbfc0827c4a262ac0b26bbca0a77bfc49124d780bd011f28d6:action", "state_id": "71ad2a6f458c24b6978521cae2e437c4f87e06564851999a3f6b43977a5a021c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6171875, -1.48046875, -1.26953125, 0.515625], "student_probs": [0.08332765102386475, 0.09553562104701996, 0.11797074973583221, 0.7031659483909607], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d8f21741068edc7e528bf72bbe0e3d9aa8744935fae5608f7b259daa4dd56824:action", "state_id": "0a94ef519599a67773433dd83222964f00921e9dd925326b9bd85d6963e40bc8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.50390625, -1.46484375, -1.1796875, 0.607421875], "student_probs": [0.08560186624526978, 0.08901185542345047, 0.11838308721780777, 0.7070032358169556], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2dec1d56201437f95e320614aa7ec0623761c76f076514937681a28607f70a1a:action", "state_id": "643e2230c3106daa7d6436a16583cdba8dc316afa59796f11355fa020e83034b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.23046875, -1.3203125, -0.44482421875, 1.23046875], "student_probs": [0.06319640576839447, 0.05776619538664818, 0.1386415660381317, 0.7403958439826965], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3b9b604043512dd81964476dc23cd91f973472c4f88d4ac7bd6e504384b3fb2d:action", "state_id": "214bbff758341822c2c6fffbc67f2dfeae9825913e02b483c2c5f0a8639a7b8f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.44921875, -1.24609375, -0.562255859375, 1.037109375], "student_probs": [0.05998785421252251, 0.07349865883588791, 0.1456352174282074, 0.7208782434463501], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d157181a5ce516cae3b64dd1dbe7156416d6fa69bd733e5cbd314a6200ac4c4e:action", "state_id": "33b03ca07336cca862ea584ab63930ec71cfefb17db04ab7193a4d378dd44aad", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.40234375, -1.37890625, -1.12109375, 0.869140625], "student_probs": [0.0766737163066864, 0.07849198579788208, 0.10157617926597595, 0.7432581186294556], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d6b81d57d56206fcde74f9abca13aab41e698314752a5450ff5c9782a1ef5db5:action", "state_id": "126099083b1d5c8b4a784756f516de7740d9f49fb126fa6bf7583c82c1de7f93", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.33203125, -1.25, -0.927734375, 0.9296875], "student_probs": [0.07585347443819046, 0.08233816176652908, 0.11364736407995224, 0.7281610369682312], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f976df0acc15dea5d28db38e28840fbd64f5d54360757ff5d0c59ad69bc086a9:action", "state_id": "48cdba58f09ef226cde7e51311795d912af5d7c70c56eb98eccb9bd42016d18c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.390625, -1.3125, -0.95703125, 0.943359375], "student_probs": [0.07172045111656189, 0.0775483027100563, 0.11064974218606949, 0.7400815486907959], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9791d0398724d96e0388c7b4cc58e99ff908d3f1f099000019e3d0894af0a489:action", "state_id": "62fab2723b1b91369de6e3aff9bc00f04eaa69883ccff68ac92fe5318e4e9060", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.017578125, -1.390625, -0.166015625, 0.97265625], "student_probs": [0.08811204135417938, 0.060676854103803635, 0.2064734399318695, 0.644737720489502], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fe1411231483b1c466f99847bec3e10425ed48e99c962a072e7140cbfa7e6e73:action", "state_id": "b873c3f87faf8f3a4f804346588bea339358a1d4d3f8798931b282a03efcc127", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.943359375, -1.2265625, -0.033203125, 0.970703125], "student_probs": [0.0907551571726799, 0.06837192177772522, 0.22550031542778015, 0.6153725385665894], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "509fa79445c1ff3b7528a46a0e0ea1b5ed105de96bab771f23527595c61c31f6:action", "state_id": "9ef2487caeaa4bb37bb622243cfc0bbcbdc405d7cd5297e391ad31eac15f6bc5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.123046875, -1.21875, -0.294921875, 0.623046875], "student_probs": [0.10070570558309555, 0.09151467680931091, 0.2305176556110382, 0.5772619843482971], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7178741f4ebd130fcc4aca336daa070b5aeaca5c3af11e8950b342075188ac61:action", "state_id": "c00930756d1c3d641f6048b904b96b15eca9c1057e400523cbf4aa2a5cd83883", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.09765625, -0.876953125, -0.15234375, 0.740234375], "student_probs": [0.09005892276763916, 0.11229926347732544, 0.23177722096443176, 0.5658645629882812], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c25d69eab4166838a4ccfe576197d84185f9dd004827274b356f82c01e9b6644:action", "state_id": "9c3db393274953dd20123016b0a5f4b3de8bf9e5d2ce403dfe1ddf694847890a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.951171875, -0.951171875, -0.072265625, 0.78515625], "student_probs": [0.0991598591208458, 0.0991598591208458, 0.23880314826965332, 0.5628771781921387], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9f839872ec1b276b55db5bcb9a44cce7039405e7e4d100dbf30fa19d43053154:action", "state_id": "ac76756b8fcee681b7eb1cf00b9e864ce4a1f36831261af8ce684dfb77d661b5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1953125, -1.4609375, -0.502197265625, 0.64453125], "student_probs": [0.09938167035579681, 0.07619857788085938, 0.19875699281692505, 0.6256627440452576], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cdbef1dcd0b3bc519bee7a59197e6fb123c9f56e55fe544c3ff817c9e06ec3a4:action", "state_id": "3623054d02835d0430c77591bb3583c98e8b6b8bbad6a3bca9cac0ff4c12eb36", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1640625, -1.22265625, -0.45263671875, 0.654296875], "student_probs": [0.09860256314277649, 0.09299107640981674, 0.20084290206432343, 0.6075634360313416], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2d50113328f04ad9b16270f49a0cc0710ad10194c5a388bdb6f14358a21a1546:action", "state_id": "a6ae79dc1f152e855b4afacbcfcdf620707b85396aac40250bf6a7c7dc2a1e82", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.24609375, -1.1953125, -0.471435546875, 0.595703125], "student_probs": [0.0949685126543045, 0.09991568326950073, 0.20606747269630432, 0.5990483164787292], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b64ee171de313ad06f641bd4866baa4d477100250907b67c3ef3ae0d5d055111:action", "state_id": "b59f1aad88a57ec9e83da0cc6bebbe5436246b7cd8d97b13cf4321ed512cbe8a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.15234375, -1.01953125, -0.396484375, 0.7890625], "student_probs": [0.08896781504154205, 0.10160443186759949, 0.1894516795873642, 0.6199761033058167], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3bc3c8b734c2115e1dfc0e2ca539ff40c835f99c55c106134f111b93152108be:action", "state_id": "2fed2356514ee850e42fc2bbb6104621f64e5b288865a8693a4e6fc53a19d781", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -0.9375, -0.296875, 0.951171875], "student_probs": [0.08646131306886673, 0.0960785299539566, 0.18232500553131104, 0.6351351141929626], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c8ceb0ff2cba0ab43b66a0ccb2121fb2df37dbc2aca3c12b06c357ca2d6d2386:action", "state_id": "4fcf2fffc7c05f30665bbe1318ec675032443d90b159007e26d59fefd2e208b2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.052734375, -0.955078125, -0.3232421875, 0.869140625], "student_probs": [0.0908234640955925, 0.10014047473669052, 0.18837033212184906, 0.6206657290458679], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1d0e55736a6382fbab5cc8cdc127399599a7f27f904fcc50b85292479bd81ca0:action", "state_id": "b67e0cd3eb8769209a9dbf9931fb0a0b5e3874163f6e305500000f6f6b806c68", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.056640625, -0.990234375, -0.4775390625, 0.861328125], "student_probs": [0.09380675852298737, 0.10024759918451309, 0.16739201545715332, 0.6385536193847656], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5c3f245bc3f3194ae6c35ab28ec3480147a2e48f62bf5db447e4eca87b1a696b:action", "state_id": "c186f3bdcb60d72b390cb0166a7f186b6d5344487949140b0cb214fcb40b1ea8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.546875, -1.3203125, -1.1328125, 0.572265625], "student_probs": [0.08270467817783356, 0.10373491048812866, 0.12512817978858948, 0.6884321570396423], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3bc7f502d72b5a8d1c6e6d9158ec411b6f55bf31a647010786fae5f72d5d3ba0:action", "state_id": "39281348c68f1efcfe7a12e28896777d300d0feeff44c6f8a4abb5f3490cbdb9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.62109375, -1.40625, -1.3046875, 0.373046875], "student_probs": [0.0912596583366394, 0.11313170194625854, 0.12522538006305695, 0.6703832149505615], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2ae8a67951a5ca4c816d642fa463ee6b982fd776789eeabd20e63999f4946f0f:action", "state_id": "a5adc25a0e0abf106d458f25eddde4a8e2b017c3cdbf3252d7539db39dcdafab", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.515625, -1.609375, -1.2890625, 0.900390625], "student_probs": [0.069609135389328, 0.06337983161211014, 0.08730941265821457, 0.7797016501426697], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f7718f22da87fff7aaf5b49e6b5d0f1c25a883b9b3dad4293fc63167fcfb684c:action", "state_id": "735e6d5bb07f3f515fd6d2a902e049591ea255831e925440f685ad3db83ec648", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.359375, -1.31640625, -1.087890625, 0.943359375], "student_probs": [0.07486538589000702, 0.0781523659825325, 0.09821666777133942, 0.7487655878067017], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e812b50e0b13569bc37ad50bf6ba73b78536e8a91649fb1ba6c02b6173106fdc:action", "state_id": "fd60fd81dfe52bce28f8f154a7f20055128b9ca982af92b92b825c907b403e69", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.31640625, -1.38671875, -0.96875, 0.884765625], "student_probs": [0.08075297623872757, 0.07527004927396774, 0.11432566493749619, 0.7296512722969055], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "35dc6f9d105409e8d0b9a714a2ca175d82ecf95e58bcaf18c072dab56160a078:action", "state_id": "128b0c9efb691ea5c21a3b43d7bc2e02082cccfe8812fc174c4668206a2aa8d9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.23046875, -1.2265625, -0.810546875, 0.94140625], "student_probs": [0.08129765838384628, 0.08161584287881851, 0.12372223287820816, 0.7133642435073853], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8fb7891c2a1ba123a74ca7c42ad2ee75b26bbd70faa3cc2bbde7a86455e012c5:action", "state_id": "b9ba2e8ad6451f10da2d4d1099a3884742fd5432386b887a2c382a505b9c940d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.080078125, -1.072265625, -0.6162109375, 1.046875], "student_probs": [0.08342147618532181, 0.08407575637102127, 0.13265781104564667, 0.699845016002655], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e52616078c69759c9e21320896c1530e5fdf5b2038268729add595212f75fbce:action", "state_id": "2807b48c6a36697e2ae1204cfdbecc76c60857d58c27c1dd56990d86ff4fb0db", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.048828125, -1.052734375, -0.5396728515625, 0.841796875], "student_probs": [0.09724321216344833, 0.09686409682035446, 0.1618015468120575, 0.6440911293029785], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8a7d2f444e77d664b7768d6ac22766af76cdfc6b8ba24849894597a191c165c3:action", "state_id": "599881dcfe8d1034a86bc49a92930ba14d2c47e5f71d0f6fb1ebf633c31de720", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0546875, -1.021484375, -0.5322265625, 0.861328125], "student_probs": [0.09511338174343109, 0.09832445532083511, 0.16037753224372864, 0.6461846232414246], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4f0b392c53cd86e2631beeba74f61ccf81f1a43c1b6324dd0c9744883f81e1b5:action", "state_id": "16a493094d66b47150bc1b1ee50c0a35226cb95572a9b8caa50e44919df2ad98", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.064453125, -0.994140625, -0.494140625, 0.884765625], "student_probs": [0.09203983098268509, 0.09874431788921356, 0.1628018468618393, 0.6464139819145203], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "32e01cb8eea83ffc50fb4a25d40a745e5c69ad5e520dad7c980fdbfbb5c791a4:action", "state_id": "10f3f237b2530836d961f89a2d797954db755b6d220ccdfc448da041dcd04eb1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.087890625, -0.97265625, -0.3369140625, 0.828125], "student_probs": [0.09062042087316513, 0.10168847441673279, 0.1920308619737625, 0.6156601905822754], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9efb4f5f7d3a1bfca5a4cc4f08f13a4b196306573c523878388113ad818e46e6:action", "state_id": "8eb4e9c752a02522d3ed6160218f9e40f96aac2e876bc6deb5ebd68a15e40440", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.076171875, -0.98828125, -0.3369140625, 0.779296875], "student_probs": [0.0945095345377922, 0.10319199413061142, 0.1979389190673828, 0.6043596267700195], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5f8e7d59ff288fc6b914bb14fcc6462dcccab0c972de4c63da1ba3d27c054257:action", "state_id": "e22e845335641aba4f2af4dbb5288660aa7e6a7f59c3535ba91cc37e256b7c52", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1328125, -1.021484375, -0.44189453125, 0.830078125], "student_probs": [0.08902193605899811, 0.09950530529022217, 0.1776474118232727, 0.6338253617286682], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8c2be45e3211140a08f4e864bb82806d3c29f4436630252335914f251a5d12bf:action", "state_id": "35b5efa225e10d3e89a2be6d8ad3832d8c069fc5a78b78f5ec4afa0e8b56e021", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.14453125, -1.044921875, -0.3779296875, 0.78125], "student_probs": [0.08994678407907486, 0.09936775267124176, 0.1936049610376358, 0.6170805096626282], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "46064e44a0a3d7c3669a52f8ab489f0caba9a726e3a8e0ee91bb428613156f64:action", "state_id": "007fd581adc3d7004374818e957ef3565da2d2565e67491cd566d499eea0d004", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.138671875, -1.130859375, -0.3173828125, 0.69921875], "student_probs": [0.09465625137090683, 0.09539864212274551, 0.2151942253112793, 0.5947508215904236], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f54850bb3e338074fa973f23abcbf51d208b17390cfd2b462eb4dbf3581bc352:action", "state_id": "8b0087773985f99ebbc17126a29e6bb16bc41cf3d462eeb69c6d00c3be3b7686", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.24609375, -1.12890625, -0.40576171875, 0.63671875], "student_probs": [0.09079824388027191, 0.10208721458911896, 0.21039189398288727, 0.5967226624488831], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "766767eeb2629200121ba705e6cd3b6cb78382f4749a8c6dd021d128fd438664:action", "state_id": "a0b2cdfe40ee066060163e1a7453ad7caf13591d1048c9656ae8f0c7e7f78079", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.23046875, -1.09765625, -0.4384765625, 0.62109375], "student_probs": [0.0932871401309967, 0.10653726011514664, 0.20595845580101013, 0.5942171216011047], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e2f004a9e3e3a1f4a6f479a723a61264fafa592600889d76cb615cd2a5e2bd47:action", "state_id": "cd1215e94373b3535aa6bce1b4d924de04cb0a3ad12705156ae09498daec9232", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.16015625, -1.037109375, -0.3896484375, 0.74609375], "student_probs": [0.09074728935956955, 0.10262950509786606, 0.19609248638153076, 0.6105307340621948], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6bf5e07f645061f55992036f8f4909a5441e0e0c43bc8e314d330361547df899:action", "state_id": "795308d57d89fe8671ad08edf576b45e3cf5281f56da83567bb5ce6312b5801e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0703125, -0.990234375, -0.2919921875, 0.8671875], "student_probs": [0.08926532417535782, 0.09670752286911011, 0.1944030076265335, 0.619624137878418], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7192d2cce54be3b8a947a903df38d13bbe7f98ac31e5fa9548fdc2459572b385:action", "state_id": "a99b1ac2df4aed2d9add894c40b0e46a9eddf0ddfda365844caf0f83d5a26ea5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.087890625, -1.025390625, -0.3515625, 0.818359375], "student_probs": [0.09190702438354492, 0.0978345200419426, 0.19192518293857574, 0.6183332204818726], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4ec4da4ae0c34f56cdee47bbcf2c35092a97f1d98d672137e7a6569dcb750ae5:action", "state_id": "8a4520e8a9946b72a63abb94a648b5899362eaaf69ba260f4b937e77c3874513", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.16796875, -1.06640625, -0.43115234375, 0.771484375], "student_probs": [0.08967709541320801, 0.0992635041475296, 0.1873599886894226, 0.623699426651001], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dcf5a7cd02fc6233d446fe2c96d234a4bd3eea62d609ac8231f46d17d088e304:action", "state_id": "704104676c0123ca49e7d8e226863561af01d8c9445e7ff9c96dcc8129971ec4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.13671875, -1.046875, -0.3857421875, 0.7890625], "student_probs": [0.09030504524707794, 0.09879402816295624, 0.19136258959770203, 0.6195383667945862], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c5f820c3d802a8a3e5f35de64819f5d1bb956f1c7d7feae1c65ec55276e6d549:action", "state_id": "5faac44f40a4f73d05c42c7d04e7cf95a82f74eefb90554a983d3f4d86987a24", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.138671875, -1.05078125, -0.5762939453125, 0.814453125], "student_probs": [0.09176503866910934, 0.10019537061452866, 0.16103298962116241, 0.6470065116882324], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "47f6790d36b71b40257104015ec6fff6f728e698c55c02805c92091df7adf2ab:action", "state_id": "0dbe9e8ae3e0296bff8a9a73385136c961ebfaa3f4ce0baa2e046fbab8249867", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.18359375, -1.060546875, -0.62841796875, 0.787109375], "student_probs": [0.09050671011209488, 0.10235743224620819, 0.1576850712299347, 0.6494508385658264], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "92353eddca34ee197a931271e798f979309e3a17d59393dd69e7408d1e23886d:action", "state_id": "24ba92de1d670b178490c1b2e44dc784cbe40036d37a64b7a33cddbc71e77fe9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.19140625, -1.0703125, -0.634765625, 0.8125], "student_probs": [0.08856213092803955, 0.09996280074119568, 0.1545233130455017, 0.6569517850875854], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c7263a4d47a4b3763185909786fd93435458dd1f3c5be3c660fdab9aacd5fb73:action", "state_id": "188324fa912a8882ee0542c5392bc65d2b8eba6e09c4bcc979f1d5a84cf764cc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.091796875, -0.9921875, -0.596923828125, 1.0546875], "student_probs": [0.08130240440368652, 0.08981795608997345, 0.13335952162742615, 0.6955201029777527], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6cc4e606fc072d3df55f8bd765523bbf4d40357d71b9b15e2f1b3077dbbed8b2:action", "state_id": "8b5bba2c315b23e78f2678d7e901f5b27db155ab7ba7f558f3d0a8a4017b9bc4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.083984375, -0.998046875, -0.580078125, 1.052734375], "student_probs": [0.08185645937919617, 0.08920210599899292, 0.13548670709133148, 0.6934547424316406], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "56ca48f632822a09d1a0121a47083c6503f9a088a0a283c61fdadc0ed797993a:action", "state_id": "4696ee52cda05ee05824cc2a7218d69600258ca98f8acfd9aa055b13eac6d39a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.48046875, -1.29296875, -1.005859375, 0.78515625], "student_probs": [0.07434523105621338, 0.08967746794223785, 0.1195015013217926, 0.7164758443832397], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "00de7cd7e422ad1de6ba8f7423962314c190fdd0dad8d82744a6ab4fd1ae5deb:action", "state_id": "0d786fd73d3c4ab221eda0af4326dde3a17d95b3785f735e3b63142fcd1b3e41", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.45703125, -1.328125, -1.0078125, 0.810546875], "student_probs": [0.07484661787748337, 0.08514426648616791, 0.11729118973016739, 0.7227179408073425], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d3f8c0ac25662fefb54eb099f5e77c6f6eaaf8211503ab45409fd422de52bfb8:action", "state_id": "c952b1d047904ceba1270c574cbafa6c3f66f9765e771c568bbc51fa57e0e249", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.484375, -1.26171875, -1.044921875, 0.73828125], "student_probs": [0.07672788947820663, 0.09586314111948013, 0.11907082051038742, 0.7083381414413452], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "08c4472005020fbffc14d35961bce2a2b9f8584401a9f48e853ff2c0b0fb64ed:action", "state_id": "0444284cc956b7eb1e748f19dc1e1dd86ff83c629d8788cd63afd86e8a00eab3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.49609375, -1.44921875, -1.1796875, 0.75390625], "student_probs": [0.07747184485197067, 0.08118979632854462, 0.10630591213703156, 0.7350324392318726], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "79c949d24493047937d47e641ae6aa834839161dae71144ae3373bea00728abc:action", "state_id": "f0c8d339bdf7bfb915b58a12f50487c6d0550cab194226317a6e8860c2b6bc1a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.65625, -1.74609375, -1.5234375, 0.47265625], "student_probs": [0.08724649995565414, 0.07974975556135178, 0.09963863343000412, 0.7333651185035706], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "34cd8a3725f5f6ef48dbe4bab2f78b0a971df7ea653785e430940e9a9fa593aa:action", "state_id": "f8248be5a1f8eaee1262fd50f2609a98dd8563dfadc77a635f91a443f5980787", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.35546875, -1.2265625, -0.7841796875, 1.09375], "student_probs": [0.06456849724054337, 0.07345205545425415, 0.11432162672281265, 0.747657835483551], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "035690c8ef2784bb69caeebcac70001291f7582658ce80ccdf0c79cea09d816d:action", "state_id": "c08c1705429a5f1e47e86a9dd4f92cfa9a3686e5cdf0659e97d1b2a971d2868c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.36328125, -1.28125, -0.8515625, 0.90234375], "student_probs": [0.07467818260192871, 0.08106239885091782, 0.12457484751939774, 0.7196845412254333], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "415358d6bf1c68fe6aab378e13a9cd187005063996ee3cd240bf516caa4ddfd4:action", "state_id": "458c6f674ce5253aa739f09febff9c845f6d07d4a09f9969aad2b3f47cff724a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.38671875, -1.15625, -0.7724609375, 0.521484375], "student_probs": [0.09217914938926697, 0.11607107520103455, 0.17037327587604523, 0.6213764548301697], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "43f37a78db7567b9bc0d3eef8e56201f4718f9eb89257193258833191ba72f9d:action", "state_id": "be09788e16242b30ddc4bedaddb89d3ce388f195830db0df1929f7ff719511e6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.087890625, -0.970703125, -0.169921875, 0.787109375], "student_probs": [0.08969134837388992, 0.10084268450737, 0.22460494935512543, 0.5848610401153564], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aee9601640720a43b0b3afc57b1d570fd0782f40cefb314291c12bf7134250e1:action", "state_id": "b26a7a84f9a219826ee207533caed05b5fd9827cc636fa38484dd6de3fc9071d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9921875, -0.98046875, -0.078125, 0.791015625], "student_probs": [0.09564705938100815, 0.09677451103925705, 0.23858542740345, 0.5689929723739624], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8b8e73dd5365f8c79807bfe4433c6e42c48b1ff1d9a683e3251a5ad76c5b9e68:action", "state_id": "3d068a2528163d57efa8fe6d7af156b0d421cc3f5b2d1089de8525c7b846c2b8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.11328125, -1.41015625, -0.240234375, 0.5625], "student_probs": [0.10548190027475357, 0.07838749885559082, 0.25254419445991516, 0.563586413860321], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5e54fdd16b194b81af063c307fb3fb80e32e7bb9540f826667bb3efea3a9fb1d:action", "state_id": "dae4110d5b8bd680b398ab5f72b12ebaf6e8e07b81ec8501c501fe1eec754d97", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.984375, -1.041015625, -0.041015625, 0.845703125], "student_probs": [0.09304141253232956, 0.08791795372962952, 0.23898577690124512, 0.5800549387931824], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fc6b9077f8aac2d46fc49c83dde6419490b844c4ed26a1939ab7bf1fdeb2514b:action", "state_id": "4377bc42c61bdb064b4cabbd6bc28d85ffe26983fc8a8f66f08ca109c44d03a7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.044921875, -1.037109375, -0.03125, 0.78515625], "student_probs": [0.09092629700899124, 0.091639444231987, 0.25056570768356323, 0.5668685436248779], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "468b509359d8a4901321790cfa7e125bb2868dc957c99ef726ca3d7eb7ad5ad0:action", "state_id": "c8647e9ef46458d55b6cd04c6d5fe429fdf1ecf9ab964139c1ed582a1545050b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.94140625, 0.01171875, 0.814453125], "student_probs": [0.08925211429595947, 0.09707166254520416, 0.2517847418785095, 0.5618915557861328], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f56c51cecc010a4b0a77d9e70f6b431a722aafc14138bcf0e8348734f60081e9:action", "state_id": "e470b3138f0d37a1e4ede058650af609de97546881660186c06ee7ba1d4c0fa7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.001953125, -0.9296875, 0.029296875, 0.80859375], "student_probs": [0.09096449613571167, 0.09778144210577011, 0.25511622428894043, 0.5561378002166748], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7d749bec139e4e4b19b82272db212e064d3afa3e7e80a28cb9880842b396b34a:action", "state_id": "df5734dcd175fcc0b81c06acb1b236fbe38f4e82483d229b27db18dd2cfdc532", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01171875, -0.919921875, 0.009765625, 0.814453125], "student_probs": [0.0902240127325058, 0.09889833629131317, 0.25058043003082275, 0.5602972507476807], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "180251dd55cccb4202a99ef62fd939d576979d0081c5394692655b4ffbfb8e27:action", "state_id": "406f35d9be8d79d17c318164075e3c4413259b4ef36ae7b3e1d74d7ab85fbd5d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.98828125, -0.91796875, 0.015625, 0.81640625], "student_probs": [0.09191315621137619, 0.09860841184854507, 0.2508237063884735, 0.5586547255516052], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7e084973edbfb7435061cdd613e5fd1fe211f69be87cd2b6768997f387ff6136:action", "state_id": "785359f1a2320860fbb0bd89e84a6c48c1b9bb47a6081049c6d56829a66510d2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0, -0.921875, -0.017578125, 0.81640625], "student_probs": [0.09172718226909637, 0.09918073564767838, 0.24499569833278656, 0.5640963315963745], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4702cebd67b93918b68b9ed51e4055844e8839ab8ccc35ec31430c5ff30835e8:action", "state_id": "c19613be672c6906f595521208fab8684e5c651c8acefa5647db83b68108cea7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.998046875, -0.921875, -0.015625, 0.806640625], "student_probs": [0.092351995408535, 0.0996614620089531, 0.24666452407836914, 0.561322033405304], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1bc4d3b02b34fb7ca93bf684175fef88a9b88ce145e42a592bd672fd67c1ffc2:action", "state_id": "0f91da507ee26461598ea4786d41030ce27ac07ffe9c8502e5b4a845b510b5c6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.013671875, -0.935546875, -0.021484375, 0.791015625], "student_probs": [0.09211108088493347, 0.0995958223938942, 0.2484353631734848, 0.5598577260971069], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dde44bc09f22c0e270f2981b9e99afd5353606b2598145d6e3ea16a56faf712b:action", "state_id": "2e86310c7f7ddab0d24e8d99f336345605b51b11033cff0be1b6843beb04bd84", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.021484375, -0.943359375, 0.0, 0.806640625], "student_probs": [0.09024634212255478, 0.09757956117391586, 0.25064244866371155, 0.5615316033363342], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "35e65f2f643ae8b8e0fdc0cd17d8c996c5adf99519ae0a73b8de4283f8896cd2:action", "state_id": "b607d900c671dd6a15956d030c01e489ce5c396070e0f6b2de28429db29bc131", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.021484375, -0.953125, 0.015625, 0.78125], "student_probs": [0.09125741571187973, 0.0977138802409172, 0.25744178891181946, 0.5535868406295776], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9d44ffd5ac4046500be371887ea2999f67c8c7bbaeb5f02198ba0972270a480c:action", "state_id": "eabb11f235eb5f64e53f618f4024c740f380a40bcb7ba2377190733d1b35181f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -0.953125, 0.021484375, 0.796875], "student_probs": [0.0896933376789093, 0.09679238498210907, 0.2565125823020935, 0.5570017099380493], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e103bd6442882e2293190894dafeb875097a7490d87659fbf207136e603968c5:action", "state_id": "94aeaece1454bbf4d71641b6a53b6e351f20979cf0e57a3df8743723b76dd9ab", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.037109375, -0.951171875, 0.025390625, 0.7890625], "student_probs": [0.08933833986520767, 0.09735539555549622, 0.2585090696811676, 0.5547971725463867], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0a5e5e5211d2c258f47764bd7e71dbc173e86a21ab9e16ba786c815dc782b961:action", "state_id": "6eb8d794460d6957f833cc54525e94c36afdb5fee1ac7be06a696c31c12dc9c0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.02734375, -0.955078125, 0.04296875, 0.7890625], "student_probs": [0.08975895494222641, 0.09648557007312775, 0.26176321506500244, 0.5519922971725464], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "46ec787c572fe1c1a42a2d9845f6509bd371f6af604e53add34c9d2bd3e1dfc4:action", "state_id": "010f90c214fb0d5c5a2b60f9b4fb820b59c83f4fa008de4083d0b3c4c2f96b0d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.046875, -0.9765625, 0.03515625, 0.796875], "student_probs": [0.0881546214222908, 0.09457609802484512, 0.2601149380207062, 0.5571543574333191], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "59e2b3f34fa86c9ebbd974b68f4e39617e728296fd8a423fae2e396fb4eb0581:action", "state_id": "7b95bb411c1f398a11aabde9e10d0b0cf4c55bf37462116256050a6d0e40948c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.037109375, -0.966796875, 0.056640625, 0.8046875], "student_probs": [0.08798050880432129, 0.09438930451869965, 0.26266127824783325, 0.5549689531326294], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e5ff256b379ee976856c790f06ff51405cbeb49d84d536e82047bef562238db1:action", "state_id": "be2d4e2024e1a4d524bfef2196c76706a152dde154291e42334f8d8433678544", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -0.958984375, 0.025390625, 0.8046875], "student_probs": [0.08878902345895767, 0.09581650793552399, 0.25641825795173645, 0.5589761734008789], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7fa91d53a51375227149ec53bf1f13cfe5450feca32b509cbdef45d501e43e8f:action", "state_id": "739f16374d26827ecdbe55a60eb4830498543d21283c7a2bbeb7b88540bfeea8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -0.955078125, 0.03515625, 0.8046875], "student_probs": [0.0880613699555397, 0.09596383571624756, 0.25832173228263855, 0.5576530694961548], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f5ee53b75b55eaff7083b113f9cd247964eb7995b4d338c1e2efc869d8d86735:action", "state_id": "e7bdac24a5b624aeaa4a4adedd22e269c396e7fff5ccbbe8c9d7cfadfe96b88e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -0.95703125, 0.029296875, 0.796875], "student_probs": [0.08875398337841034, 0.09634153544902802, 0.2583273649215698, 0.5565771460533142], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "225932275a2d0119d7fda4e11891efb06f5e5d9b0a97567890f51f4b3bfc8ebf:action", "state_id": "bf9c8c7f6776a30fcdb38ada6e3adaac954cb849f7dbbf154d936e47e235d9a4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.048828125, -0.974609375, 0.04296875, 0.787109375], "student_probs": [0.0882793664932251, 0.09508061408996582, 0.263039231300354, 0.5536007881164551], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6d1702b10168117cc4c3c66c247ab541b7faca85307389e04e9ec229bdc82a9c:action", "state_id": "ffbef194d9314108cebdc9b96cc524a0303384cfd3a60ea41677204f1a06a9a8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.072265625, -0.96875, 0.03515625, 0.779296875], "student_probs": [0.08691591024398804, 0.09639522433280945, 0.2630549669265747, 0.5536338686943054], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "51ea28455bd855625336c4951ed80d4e1cb018189e885344a0f2d83a8e4025d3:action", "state_id": "88fe45317bee53ffae5d550a468e0b8a9036dc6aa695133a2ef63adde6e401a1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.060546875, -1.005859375, 0.01171875, 0.76171875], "student_probs": [0.08957314491271973, 0.09460809081792831, 0.2617320120334625, 0.5540866851806641], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c736492f96ee7f718232ce68bc54e18745b26b0aa7549093b164a02262f32644:action", "state_id": "6fd7b6c9cbb34f1126b9700e11b23f73e9607ef450a0a6da85cfd0bf8ad5a471", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.046875, -1.005859375, 0.00390625, 0.794921875], "student_probs": [0.08920911699533463, 0.09294416755437851, 0.25512778759002686, 0.5627188682556152], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81598a1ced26648f210e97ebbf87ebf18ca5011c443a2a63428a9473690f8fc6:action", "state_id": "a1e134f892e4c749ad5a5b305ddfb0e141d54fa9e6cdf29c286be5bb4a867b0c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.060546875, -1.091796875, -0.0546875, 0.744140625], "student_probs": [0.09275095164775848, 0.08989730477333069, 0.25360485911369324, 0.5637469291687012], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ca069864b4c1bac97c58ffad8c816ee4a07008d0402a652c010e0e18f9619379:action", "state_id": "b8985727bb0e672a498a2a6d0166803189ed9e709c476aefc0b7b8c9ba4c42ff", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.02734375, -0.876953125, 0.01171875, 0.80859375], "student_probs": [0.08881102502346039, 0.10322399437427521, 0.25103020668029785, 0.5569348335266113], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2246ffae5961714879f4c6f1c3704b117d1a39afe3b256e2a16649288c0dc399:action", "state_id": "faf1dd1629e818cd48fb277064a32cbde894b94c0cfd28975be195889335e25a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5546875, -1.65625, -1.37109375, 0.77734375], "student_probs": [0.07460575550794601, 0.06740067899227142, 0.08964087069034576, 0.7683526873588562], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "efa848aa0eed381a97288cfe6cfa3ba808e1d91a1e0e9ff03dc22fb2ca58167a:action", "state_id": "5dfd35e480c7c1cd47af3960bafe20de31e482fe61bcbab55193002b0a920ce9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37109375, -1.33984375, -1.08203125, 0.869140625], "student_probs": [0.0783548504114151, 0.08084210008382797, 0.10461745411157608, 0.736185610294342], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "80d530b4c154c8c3136aaa7ae48b171a5ea61611c738efe879f8340e294eefe2:action", "state_id": "b6aacaea48f38320caa6c036b7d6d49e27bd2aa15d17acd27e191d151c8fce6f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.48828125, -1.48046875, -1.203125, 0.619140625], "student_probs": [0.08646915853023529, 0.08714734762907028, 0.1150013729929924, 0.7113820910453796], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ecf142d3f0286903ada847c3ee1997cd8fcbb4b64abab5b6307cc4422380a83e:action", "state_id": "a9357960d1ad5a8a48d58ce90ce2cffaf19ded817c08962f49180496643e918c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58203125, -1.47265625, -1.2578125, 0.4140625], "student_probs": [0.0920913815498352, 0.10273535549640656, 0.12735776603221893, 0.6778154969215393], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d8b98b48f77537fc84bc0ec2038af3459252fa77f5d7106ba4c95a86f04ef4b8:action", "state_id": "612f8208cbb206cae3f548e1a0eab8be3e2998d25fc280ee1f793af3fa1a6238", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.203125, -1.49609375, -0.3857421875, 0.5546875], "student_probs": [0.10193319618701935, 0.07604679465293884, 0.23083437979221344, 0.5911856889724731], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dc954a8930aba59ff3902607cf6c06d84954d9d30611f77390e07329896f9751:action", "state_id": "a9ff38ae1bb3429d5d00a76aff4b73eba1a7c05eb7141cda27b08420ce3d9005", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1171875, -1.34375, -0.2421875, 0.73046875], "student_probs": [0.09486697614192963, 0.07563454657793045, 0.22757405042648315, 0.6019244194030762], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3172c134220334b7ad51b3af964778bc99919c8a9e11a95d623c198462e7e716:action", "state_id": "72653955efa17db2dada8ebb407205aaa0304891ac1d340d45e5665cac697ee5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3828125, -1.34375, -0.5321044921875, 0.65625], "student_probs": [0.08288753032684326, 0.0861893892288208, 0.1940648853778839, 0.6368582248687744], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ff0ae4e95a1893181fcb130c04d95bac1c32d2cfe3fcd91891d61b170c64b409:action", "state_id": "781c14c9926b19cd04312310cbcb6bcef8c8e0ab121354aa10514ca67ec489f0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2265625, -1.20703125, -0.8828125, 0.734375], "student_probs": [0.09491326659917831, 0.09678526222705841, 0.1338491588830948, 0.6744523048400879], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4bb892f2a0b54fb9872a5675708d40dd6854cf021a05a16b569a46d5e9045cff:action", "state_id": "6600a225a69d3f78c0ca3849787c3c7e59ac45e613b0c38a165a93bc05e070f4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.30078125, -1.3046875, -0.912109375, 0.970703125], "student_probs": [0.07595954835414886, 0.07566341012716293, 0.1120418906211853, 0.7363350987434387], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5db1ad6d12c5a2ba69e63a75f8a535c256b2be93ee42739ce874fb20d8b900f0:action", "state_id": "47f0e3901b88f491b04665527a01f79a7f767ab39795deb9d08c8f470b34f69b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.36328125, -1.28125, -0.94140625, 0.998046875], "student_probs": [0.07034656405448914, 0.07636047899723053, 0.10726570338010788, 0.7460272312164307], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "22ef499648235ea7a1649f79df26f85fb584b4991f09df2115ce1ff0ad743a3f:action", "state_id": "0b3fe85339c8da559659f6d43961941826a588d65932521f54aff14f45eab304", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.62890625, -1.76953125, -1.48828125, 0.51171875], "student_probs": [0.08677121251821518, 0.07538813352584839, 0.09987305104732513, 0.7379676103591919], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2e4f44e91b823c2332570652e62e404b3239252a1176c545d3326c1411734351:action", "state_id": "76e1df98d21271028e6ae7c933b85098a4f2d47d9fcc54fd5d695c026bd7d0b0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.33203125, -1.21875, -0.7548828125, 1.119140625], "student_probs": [0.06450433284044266, 0.07224142551422119, 0.11487916857004166, 0.7483750581741333], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c659fbc597fc416faa091672e0544ea5ca01e049258f4193ac882617b3723478:action", "state_id": "a5bb455ec2f13a59d143a21c6948cbf7a31be38eeadd3fdd7cd7996327af875d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3359375, -1.265625, -0.833984375, 0.904296875], "student_probs": [0.0762176513671875, 0.08176960796117783, 0.12590733170509338, 0.7161054611206055], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "730f1f56721a7d5d3f0f15e317e8ab7710bc60c9e8c7d28f560aa7db75fd59b4:action", "state_id": "09b9393af170302c4548004db658f04280953f0155b27f711b1239f135cb1fa0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.265625, -1.1640625, -0.66796875, 0.857421875], "student_probs": [0.0814245268702507, 0.09012873470783234, 0.14801782369613647, 0.6804289221763611], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "48a68623821ff079991ac8d21dd3347589ad94178bbbdb8de6df3b0ccb513609:action", "state_id": "e6af37b6c19a1c6dce067fa7f424d541fdba67b6e10fd5f532895a341d00bbe9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.994140625, -1.01953125, -0.083984375, 1.01171875], "student_probs": [0.08408930897712708, 0.0819811001420021, 0.20893758535385132, 0.6249919533729553], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8699f153cdf36b851836eccaa0af3afacb76c9279f7f7614079d624ba520b8f7:action", "state_id": "b75e91fc798b875156a841541a98fd85121ad8e569237db4b201eb0b2be56439", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.978515625, -0.97265625, -0.046875, 0.82421875], "student_probs": [0.09424396604299545, 0.09479779005050659, 0.23925438523292542, 0.5717038512229919], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "df0e2aa01365270f5f8297f1da3ca83db135c600f78f841f6fc79e194cb4a8f0:action", "state_id": "623d2c43c250ca60e3236a9dc4eca70b08f86df4949ad75bc9217797c48b50c6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2578125, -1.3203125, -0.5258941650390625, 1.220703125], "student_probs": [0.06272727251052856, 0.05892682075500488, 0.13041408360004425, 0.7479318976402283], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "76cb1dbcea988de0ac9220733d5a7a83cbb02041bbc21da41ba76c852c9ba90d:action", "state_id": "2b4f1ca0739f5474236e992696090bb2c93ee9d06cbb9f7147bfe50aec0e9873", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.28125, -1.16796875, -0.39306640625, 1.12109375], "student_probs": [0.06410379707813263, 0.0717928484082222, 0.15581777691841125, 0.7082855701446533], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0a3c413848b8867441b07ce23347295ae4744609ad5a75fd6caf4cf78405e504:action", "state_id": "39b6d6134d60a0ac1ba14ef266dd4d762a6742b2d04f34180ca49b85d0aee848", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.30078125, -1.3046875, -0.44482421875, 0.712890625], "student_probs": [0.08445589244365692, 0.08412663638591766, 0.19877758622169495, 0.6326398849487305], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ac0b47516e376e8fa6dea67bbb8db8c81cc3c8cc2167ff81ac9427600886e1cd:action", "state_id": "9ccd5f7d0fd730583f31dec92790d1a779b8ad6faf08225921082a5dd257f201", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.19921875, -1.0859375, -0.271484375, 0.876953125], "student_probs": [0.07922294735908508, 0.0887254849076271, 0.20033688843250275, 0.6317147016525269], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2082d5f69010d99e1b5ea1e89b3738130a995e3c21e5f69427a7349bc4dfc38e:action", "state_id": "419807054b3a07e6b3aef8859a83cf5704ddca1b50c34da84e13300171f51916", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03125, -1.02734375, -0.11328125, 0.943359375], "student_probs": [0.08538311719894409, 0.08571729809045792, 0.21381628513336182, 0.6150833368301392], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5de51d08adb2b5ece675f2d33cc05796e6edbdbc119a98f05d8126933aff1cee:action", "state_id": "069d9c64165510cb841c69e6c55c6f57504cfba1465340773605ecd1bde343b9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6328125, -1.609375, -1.22265625, 0.767578125], "student_probs": [0.06868956983089447, 0.07031849026679993, 0.10351883620023727, 0.7574730515480042], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "df2267e3b51006e143d8a18a905ff20d2c49e0aebeafcf30765509019f9e341c:action", "state_id": "7de9f44f4c485deb1b64b5384a651393268669b05ef74254d78d26fbe35422a6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3125, -1.3203125, -0.888671875, 0.9375], "student_probs": [0.07687722891569138, 0.07627896964550018, 0.1174529567360878, 0.7293908596038818], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "beeaec5fddc216f17496285f16eb45556348090a8cf54dc604f4ac53bb1b83ab:action", "state_id": "d8a4bb7b6af0c359b12432b02ac0b16773e18423b6dbfb53e7bf4dedce160174", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3046875, -1.3671875, -0.8203125, 0.86328125], "student_probs": [0.0812804251909256, 0.07635589689016342, 0.1319311559200287, 0.7104325294494629], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e45f484ab0e94403cd514ee9e8057691aff674377144c08bc90ae68cc4ad3abc:action", "state_id": "09bb50e688dc8b52a74fa61c84466a7ea5681266efe626d06aef11ec1b804314", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -1.0390625, -0.5303955078125, 0.841796875], "student_probs": [0.09748197346925735, 0.09786350280046463, 0.1627541333436966, 0.641900360584259], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "69b2fee1b8882e7074cf37f4037e3dfe9f4ed6a8712d65ed590397e3ab5cd4c2:action", "state_id": "7476408d3338c87bf862a87a61068bedf973b8cc55abdbf85e77d445e7ea265d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.048828125, -1.005859375, -0.533935546875, 0.884765625], "student_probs": [0.09405852109193802, 0.09818818420171738, 0.15740305185317993, 0.6503501534461975], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5a2bd93178e1323e5fea5d9089b1484e4b7be76fca6a9ef194d6b071c174934a:action", "state_id": "5209d64ef2c786a24c1f5fe74c14b6c08c58d28172ac6f3f40d1fbaa8f3a551e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0234375, -0.99609375, -0.408203125, 0.91015625], "student_probs": [0.09266015142202377, 0.09522878378629684, 0.17142963409423828, 0.6406813859939575], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "13675056e09c44e28d1b2a43a2eaf9ab8967822038f28b094564ee46c31e56d6:action", "state_id": "468bf24cdaefdc045da5cb20480b1ba0272e9ed1cfe46d1f6f84695b24e74a6f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.109375, -1.01953125, -0.3017578125, 0.77734375], "student_probs": [0.09145577996969223, 0.10005293041467667, 0.20509490370750427, 0.6033964157104492], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b99775232c29200bf464cec0f9e8b065465e2fa824d540ca1bd56730ff1e7491:action", "state_id": "1b3f078ae61888079960547b0767ac7c8ace2c781634551114e002ae47051cc3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.16015625, -1.029296875, -0.4404296875, 0.853515625], "student_probs": [0.08558382093906403, 0.09754908084869385, 0.17577816545963287, 0.6410889029502869], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "626441619b341a73f854649543ba61891e7fff594cbd2f7c2b11ae8b0960e9f5:action", "state_id": "1416813b0255f143b15eb1ea5a2472acc963a564086e0c289317a36eb4302eb3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.11328125, -1.1640625, -0.6962890625, 0.837890625], "student_probs": [0.095194511115551, 0.09048110991716385, 0.14444726705551147, 0.6698770523071289], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "eb1b7752404ff43801cb5d7cdcbc46d34f2af8015eb22ef8ca4e00d9b42dadc8:action", "state_id": "fa910c24d6402e58792d014133c816f08e57c6093124930aa37b46ad122ca8f2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1875, -1.34765625, -0.6767578125, 0.68359375], "student_probs": [0.09986083954572678, 0.08508250117301941, 0.1664208471775055, 0.6486358046531677], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c95a1ad5f12fb6d21b674f37b3bc4d971b90d4501c79d55ecb75d94ed7eca7d2:action", "state_id": "0a61176d32684ef887496730f6ef9f6853e51ef37bf2bdebc8a9258ef579e9c1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.28125, -1.328125, -0.890625, 1.138671875], "student_probs": [0.0681333914399147, 0.0650133341550827, 0.10069461911916733, 0.7661586403846741], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "92118bb5d1cc33f3130e99b458b529d602c7660b84243aaca242e4b58779d7cf:action", "state_id": "0b6e721c008deec566a8e0225cb91223fe1ac003754836752203b77756d99d96", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.484375, -1.5234375, -1.015625, 0.755859375], "student_probs": [0.07718896865844727, 0.07423190772533417, 0.12334761768579483, 0.7252315282821655], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6bc64bee40214eb313ac2df861f128cf6788ab1f96d1ffb241ab05e2adfcde43:action", "state_id": "3b1d2130557aeeb8dbe4ff2527dd164d16fa90f356d615782671b227202a61a1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.19140625, -1.1484375, -0.7275390625, 1.16015625], "student_probs": [0.0707409530878067, 0.07384685426950455, 0.11249309033155441, 0.7429190874099731], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4af4bfac5a820ba548046b452d723db3186d65fe60fc8323e68244fb920052cd:action", "state_id": "364b845ba6151580b89c150888f749ae93bb5fb4eebceeca77a73cdaa96325c1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.15234375, -1.041015625, -0.52923583984375, 1.208984375], "student_probs": [0.06855183094739914, 0.07662460207939148, 0.12782959640026093, 0.7269940376281738], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5800ff8635c1e505921c60e30b5cab496fd22baea3285bbcc5a2bb68bfb9ec28:action", "state_id": "0f0b2e070baea2fc2d544db448451f80903d1587bb3a177adce9bbf3700db79b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.19140625, -1.1953125, -0.62109375, 0.708984375], "student_probs": [0.09566240757703781, 0.0952894538640976, 0.1692095547914505, 0.6398386359214783], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cc6f6da2bb66785cbefb021d3a0e40212b50271aa7cb9bac81e2fd7acb13d785:action", "state_id": "223e865992555ecf3839475832b78a9e5ac9e073b72a8c7a8ac34cd489a0e4f1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2109375, -1.099609375, -0.609375, 0.80859375], "student_probs": [0.08712682127952576, 0.09738700836896896, 0.15900366008281708, 0.656482458114624], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e4fcc4b8f864a1af6466da9bff0a15585c1022e3b04f448f95e26b0f32819d31:action", "state_id": "160bd77e8a2116ff44ada0d60ae0351260b54066b683d23cb34d5b891213723b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1875, -1.048828125, -0.5029296875, 0.794921875], "student_probs": [0.08778110891580582, 0.10083829611539841, 0.17406287789344788, 0.6373176574707031], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0a4103cfc540c29451f5c0bc233f509b519418824378c8f4ea9daf3722e2ef40:action", "state_id": "4b4bdf852f2b93efcff1f2122ba491ee635d55aa2c6a21ecf8e174fee4be1375", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.06640625, -0.95703125, -0.361328125, 0.828125], "student_probs": [0.09268685430288315, 0.10339966416358948, 0.18759864568710327, 0.6163148283958435], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "744d9d609c6dacc90c7249da654d5d895a7153f24c9dce66164eddde92e6c95f:action", "state_id": "5caeeb0627ca8b85aa5e0c5d9cc6e0b1f768f15e3f28a62ab0c6115228585dd5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.126953125, -0.984375, -0.5784912109375, 0.888671875], "student_probs": [0.08780209720134735, 0.10125717520713806, 0.15194936096668243, 0.6589913368225098], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7b3a323f054a44ec83b3c6696b0bcb68115a1099ac021df39a19d56171cebb0b:action", "state_id": "7e52adf95ac69777e40513e968d19ab9f7a9fcfd63c6f02d132f837cf0f1fa1e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.375, -1.23046875, -0.939453125, 0.62890625], "student_probs": [0.0899338647723198, 0.10391838103532791, 0.13902050256729126, 0.667127251625061], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "08e04a22bcdefcd441622f6549bc50f3a3455865cd06cd41b517848cacf3abae:action", "state_id": "aee50093bcb1fe78954f6114580bc22c32931d6d046af106719bf88e31abf543", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.39453125, -1.2578125, -0.958984375, 0.71484375], "student_probs": [0.08378436416387558, 0.09605924785137177, 0.12951456010341644, 0.6906418800354004], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c075c1200b88d786ea6fa036562e2f33df5c337b261c121e5e986ad61947b057:action", "state_id": "5d69d56d345a5932087663c6b9cbbc70bd5289861ccb49a6e60dd9ec648498ca", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.41796875, -1.265625, -0.962890625, 0.798828125], "student_probs": [0.07740691304206848, 0.09014502912759781, 0.12201623618602753, 0.7104317545890808], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c2f5a2db73f8312092e787199f700bd9badcacdc5f8259ea0cbd86adb78c898a:action", "state_id": "6e2b526c2eb5e45a5046fc42db7b0e053b3963872b74358cfc25f05c30f2758c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.22265625, -1.15234375, -0.849609375, 1.064453125], "student_probs": [0.07478632032871246, 0.08023400604724884, 0.10860113054513931, 0.73637855052948], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e339fe5578f75377988284bd13d1e376b137fc905406cf44a3c12bb3af362f4b:action", "state_id": "052bbc41f0a649b6fc27e79412a3ccbfa6b9ed6f06a16d7625592094f467997e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3046875, -1.2265625, -0.892578125, 1.0234375], "student_probs": [0.07220245152711868, 0.07806946337223053, 0.10902567207813263, 0.7407024502754211], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "73c8d9ed12c0deb7046e1d38bcd9b9446e66cad263172ce98bf140208a4b4033:action", "state_id": "afbc1b549714fea678d3d33b485adb8783119ced21d510723848af506f909ceb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37109375, -1.30078125, -0.94140625, 0.984375], "student_probs": [0.07065793871879578, 0.07580490410327911, 0.10858551412820816, 0.744951605796814], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a26f8928caac71bd340dd3eb55f27120589b8dba0c47001f88560c67423d33d2:action", "state_id": "7fb5d0ac32655bc5958a93df59b0a08342902c1846db5e41c5cb102eb9cf88f6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3828125, -1.31640625, -0.96484375, 0.9609375], "student_probs": [0.07138893008232117, 0.0762905478477478, 0.10843073576688766, 0.7438897490501404], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "14281a1c1e8fcae1279047a73c844d8ab2e426dc26637065e64eb58ce4747616:action", "state_id": "8f5d78c31c036c849d5ecb6289c52865d9626fcc6ffad51b32217b5ce62a88b7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3125, -1.28125, -0.923828125, 1.01171875], "student_probs": [0.0728573128581047, 0.07517005503177643, 0.10746603459119797, 0.7445065975189209], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8467dcf46e558d23fb3f4046bfcabc3222b067cff112cbb5f1df8792697ddc1d:action", "state_id": "f88ba6813f2aa68d848dd9205c08f53ae5831cda67c0ae62c9f228add803ae00", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.40234375, -1.3359375, -1.04296875, 0.87890625], "student_probs": [0.07524452358484268, 0.08041086792945862, 0.10778280347585678, 0.7365617752075195], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fa07af7464aba4533ba2e6931d5c762cb13cdf39ee9f8bab9eb8ab8b3fb6b6b5:action", "state_id": "f29f60c5cb5b6102bd28c6d632731b7d08776ca7996e9c849780644566463c8f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.44140625, -1.36328125, -1.060546875, 0.828125], "student_probs": [0.07564489543437958, 0.08179163187742233, 0.11070945858955383, 0.7318540811538696], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e08a0fb30e53c30270086979f81a9827d78a8e2d1d629f69125353264989ddea:action", "state_id": "72e8a2915faeab6b80ec384058e698c52c6e71a8fc293cae92eb64806afbe815", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.390625, -1.3515625, -1.013671875, 0.88671875], "student_probs": [0.07548070698976517, 0.07848751544952393, 0.11003849655389786, 0.7359932065010071], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0d27c132f40c071a3c2336a1dfea2f34f642e4195c28d260f7e0837570f1731a:action", "state_id": "055171399fc49751ec00f0161b780ee0d11dd5373e2f379ac64e0cb5acfe32d3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.359375, -1.33203125, -1.009765625, 0.87890625], "student_probs": [0.07798223197460175, 0.08014398068189621, 0.11061882972717285, 0.7312549948692322], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6e29532681ab65ec1deb00a99cd677c12846414c1fa5c0a5918d886d3e0794c2:action", "state_id": "f163fd402136ce5d9e191b51897c83fc0ea598a286358379265e3f38502abdb9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4375, -1.359375, -1.12109375, 0.76171875], "student_probs": [0.08018338680267334, 0.08669891208410263, 0.11002665013074875, 0.7230910062789917], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0b92349179d0765dc4601aad2e97a2711eb15922f3ce82a9d846ee1de2085daf:action", "state_id": "b66ef0b68269bdd7d8773f78049ea25a206d0446bf46ee64b7df3ae2fae71063", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.41015625, -1.33984375, -1.1328125, 0.88671875], "student_probs": [0.07498767971992493, 0.08045003563165665, 0.09895523637533188, 0.7456070780754089], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1b67330cc393566841bb9418f8ad009d4784441d7c7841fb7f955b461a83e73a:action", "state_id": "21a1e5b1e92da444fe8b5f37edbf6ce099395f5cca82e507910530bdf1efa941", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.51171875, -1.640625, -1.4140625, 0.69921875], "student_probs": [0.0826043114066124, 0.07261384278535843, 0.09107816219329834, 0.7537037134170532], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "93978794633dcb1392f276e16d6b3e9b784d5852fb3f7aa670e9a195ef752999:action", "state_id": "7903bc3339eb3ab2fdc6ad3339ccc152ddd6716e6995ea7240fb531d77d5841e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.359375, -1.3359375, -1.1015625, 0.7578125], "student_probs": [0.08601744472980499, 0.08805728703737259, 0.11131484061479568, 0.7146104574203491], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5a480b728d284957e1b5369a84f0ef257fb93cd95a33110e8fd21dac82e0bcf9:action", "state_id": "5c8246fe7dc90ad92b985a3babc77f3065b8ac3c35b1cb172e768a9616985a7a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5078125, -1.41015625, -1.1953125, 0.466796875], "student_probs": [0.09369237720966339, 0.1033036932349205, 0.12806230783462524, 0.6749416589736938], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7670e2f9d27969cb6424c538eadb7e8c0c8931d95c615e4a2fee888602f14a26:action", "state_id": "034e50f1750ad717a4be7cd655c03d027634e21d2d969bae3d4f9764b8ae419a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55078125, -1.3984375, -1.1953125, 0.427734375], "student_probs": [0.09239233285188675, 0.10759645700454712, 0.13182993233203888, 0.668181300163269], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ba46f5e2da1b5f2ab3083cc72cad95f91c9c7410059a9f76fddd409d0ee93593:action", "state_id": "28e8ccc86297f931155fd60e6511bd640a230d8fabf88c5ab1322c6b28766b10", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2421875, -1.32421875, -0.7626953125, 0.92578125], "student_probs": [0.08145306259393692, 0.07503807544708252, 0.13156737387180328, 0.7119414806365967], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "24ace43cbecd0be997ee131957bb1981705dcb3c62e2129f9ca390af906b9df3:action", "state_id": "61e128667991480ead31a2dc28498a2d6381c55bc2f284eee3d5f53c61f23542", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.068359375, -1.1015625, -0.67236328125, 1.142578125], "student_probs": [0.07950723171234131, 0.07691068947315216, 0.11813689768314362, 0.7254451513290405], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e43575037e6705ed25d0c0d368e248ad4baf7639dbf3dc3d1adeb243cd47dcf2:action", "state_id": "b63d90dc04c2404c15784ef51f124d28a263c91595bcedf032cd27f1ffd8e658", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.109375, -1.16015625, -0.618896484375, 0.794921875], "student_probs": [0.09710370004177094, 0.09229576587677002, 0.15857981145381927, 0.6520207524299622], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "37fcdf796c26adb889394821a5b9c5c55706d2019fd982736839669e27ab1774:action", "state_id": "175b3898e6b7e9cff00d76d17a6f6250020da90ebceb68149c82bfa1db71fb63", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.064453125, -1.025390625, -0.564483642578125, 0.90625], "student_probs": [0.09204521030187607, 0.09571187943220139, 0.1517522633075714, 0.6604906916618347], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e2fc9292efff4572af0b858b31772abc4d0c63fd51d3622ef16d531e5d23356d:action", "state_id": "2416ceab6150b669d861968737e9d357a7aa320615f0550ed7dd53795867bbbf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.09765625, -1.015625, -0.45703125, 0.826171875], "student_probs": [0.09233374893665314, 0.10022734105587006, 0.17521865665912628, 0.6322202682495117], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e582fe073a0abac531cc06db58c0a3625f339b2df6cabbb72379b8fb5bb9dd74:action", "state_id": "462b3bfa7d5476a807d78efe393794c9584dc1cb91d249cd26283cb1c6f859b4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.150390625, -1.021484375, -0.33984375, 0.78125], "student_probs": [0.08859323710203171, 0.10078220814466476, 0.19925838708877563, 0.6113662123680115], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4533fba5d66b0ef203cf63a86b690c02ba28e0603ec74c2259582881962df7fe:action", "state_id": "26657483d8b8841d6f73e8e63fc428dd2e989ebd40ad49a3a24f37e1f115c952", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0625, -1.005859375, -0.42724609375, 0.876953125], "student_probs": [0.09173702448606491, 0.09708304703235626, 0.17315377295017242, 0.6380261182785034], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d7b4b36f37e872be3c2552978255aebfd8310b11557d310f11cb634c9cd3751f:action", "state_id": "b7e00f08d8e5f83f57bddc787b006fe349e90c8b41cc1f626ab33740906cc953", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.076171875, -0.95703125, -0.5833740234375, 0.828125], "student_probs": [0.09543711692094803, 0.10751261562108994, 0.15622003376483917, 0.6408301591873169], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "72d68abde557e0197a3b73bbd165cda69228bf73aa85390fc704ee273560f092:action", "state_id": "92231a9718bbbc0583710c9b00b7cf3107f043bc13f590387b6917f69113be5d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.08984375, -1.18359375, -0.62841796875, 0.828125], "student_probs": [0.09705004841089249, 0.08836507052183151, 0.15395380556583405, 0.6606310606002808], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fff2b435aa7e5259510c1473586e34328a8de400efabb27b3271e4e1be727e0f:action", "state_id": "4db787ca3910360005175281fe20eb1e479a456da1a09c90d992e82e7d115f0d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.50390625, -1.59765625, -1.0078125, 0.640625], "student_probs": [0.0827072262763977, 0.07530578225851059, 0.13582953810691833, 0.7061574459075928], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d1d2e3ad4c15ecae81914293f4e9d8154d715921f28c7595ca38e5abd30ce967:action", "state_id": "e302d12053d556a6ddba1c6a5bd7d7499e1655c7d9ed0b2b429b00a0e045962b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.34375, -1.39453125, -0.8046875, 0.966796875], "student_probs": [0.07275467365980148, 0.0691523402929306, 0.12473052740097046, 0.7333624362945557], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "69520d9b651f4e6255123203a10822be03ceb23c5139f43fc7d5c6e4d62838a0:action", "state_id": "245e92042af38274bdc8323e4f2ff9051b46e738d4a5b2dd6f7ef9dfb394bc20", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.98046875, -1.005859375, -0.15625, 1.326171875], "student_probs": [0.06995127350091934, 0.06819752603769302, 0.15949581563472748, 0.7023553848266602], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ae9b24b035ff5831abf9689e724ae5787e9d1b234655ff5d36a3d1a5c6279081:action", "state_id": "fb26a11a1f783a65c856184fb220764361748467a1e6a2dd22b930cb1b3fade2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.939453125, -1.017578125, -0.17578125, 1.373046875], "student_probs": [0.07056847214698792, 0.06526517122983932, 0.15144997835159302, 0.7127163410186768], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5938b6a837b91be0034a5a2d7e8883b4c3e0ce6765ba1771533c8a2d1735f034:action", "state_id": "553b0f627d9125fc9036eb8aa631f35a4d2d2bc74076893aff5d2ed74ad6a042", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0234375, -1.19140625, -0.296875, 1.34765625], "student_probs": [0.06838804483413696, 0.05781389772891998, 0.14142371714115143, 0.7323743104934692], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "00b3628b4815b96cb39de5339cd7d41effb08bf15803b1c9a2b6925c91173800:action", "state_id": "087c0987946babd77ee563aaf533c4d2338e5c61f67f92841878738482823229", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.42578125, -1.60546875, -0.861328125, 0.8828125], "student_probs": [0.07323708385229111, 0.061191871762275696, 0.1287863701581955, 0.7367846369743347], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1955e40436bd80a5e2c15bba3831f4edd5d2a2613fa3acc920ebd96225617b8f:action", "state_id": "afc3e06fb4ef642bbe7cb700b6d02299355f96d58122101a057f32bc691f2598", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.41015625, -1.4375, -0.69140625, 0.9453125], "student_probs": [0.06864377856254578, 0.06679222732782364, 0.14084789156913757, 0.723716139793396], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e24c7bbf82aafa00bb5e3a98426964f53582a784bb110422ae31affe084bc1eb:action", "state_id": "2d3dbbe6dfe4fd63cb0798597a3f9b72b800e8fedcc7fe6b28b3eae9e9221ce2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.94140625, -0.98828125, -0.06640625, 1.37109375], "student_probs": [0.069191113114357, 0.06602261960506439, 0.16598084568977356, 0.6988054513931274], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e7008d6a44bb1ecccc4f25a16097c3f634af0e7f1716955a6590f115edfe780f:action", "state_id": "ff57f665fdf94965af9df0faefe0998f6ad57fa050cf6e6f5f3b2a6d9ea0375e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9296875, -0.9609375, -0.060546875, 1.3984375], "student_probs": [0.0684332326054573, 0.066327765583992, 0.16320371627807617, 0.7020352482795715], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "63946686b2ff9505c00c0892223734d0b0cb0cf07f7e956f2d61ed468dafb89a:action", "state_id": "707d4e89a46dfd58102e02d65a23445d3ebcaf728c436f04667dba12e5870644", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.93359375, -0.955078125, -0.046875, 1.390625], "student_probs": [0.06837798655033112, 0.06692460179328918, 0.1659637987613678, 0.6987336874008179], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bd2ef162342629d31af6f3803eaa57bca0f6f47f46333fa0b15c4cde9f90b2d8:action", "state_id": "365a41c105087b33b2e282b8d245735e1175184128b462b3c8af1705281212e7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.951171875, -0.966796875, -0.0546875, 1.388671875], "student_probs": [0.06749878078699112, 0.06645230948925018, 0.16543756425380707, 0.7006113529205322], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "601e20c66229fea0be08572770901bf8710e40b10992498d84df2e4b30f5f6f9:action", "state_id": "efaa77126218c032a9bf11503fd64b033c98cb7342cfaeff171a3decfc2d4528", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.947265625, -0.966796875, -0.05859375, 1.388671875], "student_probs": [0.06778877228498459, 0.06647761911153793, 0.16485536098480225, 0.7008782029151917], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9b177bce82084c929025a6d9594dd8825dd64917d2e4a09b9c82178268bdda3e:action", "state_id": "632728564beac59d91b6f47e1a62d1b37109963d0886fb6e105a42f6d6b0d8e2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.931640625, -0.978515625, -0.060546875, 1.388671875], "student_probs": [0.06885826587677002, 0.06570501625537872, 0.1645384132862091, 0.700898289680481], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bdf5e316f100e9a76c2eae130def13682b13792a22bd2a188739a6c83dde5a4c:action", "state_id": "7a5e83752102630a9cb8d43f0dd4e3a8c6bd6fbc94ec9650f8bb608c88fdeb84", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.41796875, -1.484375, -1.14453125, 1.013671875], "student_probs": [0.06836317479610443, 0.06397087872028351, 0.08986169844865799, 0.7778042554855347], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4108e1a9935a028ba36ccf18862da1d6b17d3c4827054bdd0644361dee50377a:action", "state_id": "65164d20d08f8af07006edb35e5c32cf0694b1a18871e3e115e629351b6eadf3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.38671875, -1.46875, -0.99609375, 0.865234375], "student_probs": [0.07748674601316452, 0.07138413190841675, 0.11451799422502518, 0.7366110682487488], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "70f059e66f4283d3a4647675c6deea242a4fce439ae5ed9d03b962d3e0792c19:action", "state_id": "723110a828118e126fe40041a9588a2f5d4a027f9d5a696ca842f1ccb7dd4a53", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.43359375, -1.4921875, -0.990234375, 0.525390625], "student_probs": [0.09440069645643234, 0.08902833610773087, 0.14706988632678986, 0.6695011258125305], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "547eb5fbc53840b5db326beea6d74640ac991235fd136f31d25a4903377be8be:action", "state_id": "be729b28b0e881886fc1edd8d25a27aa199f807978dc587e119fae0b2ac27b7f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4921875, -1.37109375, -0.978515625, 0.57421875], "student_probs": [0.08549536019563675, 0.09650123119354248, 0.14289841055870056, 0.6751050353050232], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8f227095cd4dbc97b787acf54533492ff0c35e1fd5ebc4acbf54886120ce17ca:action", "state_id": "d12504e416de0dbde6eca1c85d67d695e8349efae06759f7591bc3b9354c2165", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.17578125, -1.06640625, -0.4306640625, 0.79296875], "student_probs": [0.0878426805138588, 0.09799559414386749, 0.185057133436203, 0.6291045546531677], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "319c5020d3b2984a0b09e5cb1f7e64d60d09130fff84565d0a0001714162c276:action", "state_id": "6ca46d485756220df65a69f81e8c53063f0245371b87d32a2201d71cc7202f72", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.146484375, -1.001953125, -0.530029296875, 0.853515625], "student_probs": [0.0877431258559227, 0.10138699412345886, 0.16253098845481873, 0.6483389139175415], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "69d49c920a706e353b830de2df8d264a7ce942b30f865ebdfb50c30e0b503a88:action", "state_id": "c9c840f3c704b086e918c5fd83e1416e461ed803f782323e397cfdfb9d866aa0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.53125, -1.328125, -1.05859375, 0.533203125], "student_probs": [0.08539372682571411, 0.10462658107280731, 0.13699287176132202, 0.6729868054389954], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b75ae468f11acd5c8795a714f8f6fb731d6acfd662727bbad9b203dea7ccec72:action", "state_id": "1eee6cb9193e911184c0b6366496f79caf91837ccdade2e586b528c534e4398e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.53125, -1.3984375, -1.060546875, 0.5859375], "student_probs": [0.08298139274120331, 0.09476771950721741, 0.1328631341457367, 0.6893877387046814], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b9e64c44b4a7f0bdcb44ea6ae1c71ea915d293094d4a078523bd5040c3adb7a7:action", "state_id": "3aab338e325360cd1ca086c6bc83b199ea6eadc72cc6d6876cc6ec6c5937c6c8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.43359375, -1.30078125, -1.017578125, 0.890625], "student_probs": [0.07206355780363083, 0.08229916542768478, 0.10924183577299118, 0.7363954186439514], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3a7f8c77470105f0f94378436f857426dafdce13960b81fe1890bfadf9400750:action", "state_id": "1d3544df04e8f83e9366a439c62e40f32d96bc237a4b7801faf9be41cde8c3eb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.390625, -1.328125, -1.0234375, 0.90234375], "student_probs": [0.07455754280090332, 0.07936608791351318, 0.10763636976480484, 0.7384400367736816], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6947a58a3858c3b353ad4d91cf458aa449e90d7f4ec1ffd8e4cf3a02a0f7d7db:action", "state_id": "bba246c728659ba6ae4db2348edd7e3aa8b80ac2cd77fde2b71990e8a90ad562", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3984375, -1.3125, -1.01171875, 0.966796875], "student_probs": [0.07038082927465439, 0.07669667899608612, 0.10361060500144958, 0.7493118643760681], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "350eb9d6195947e7d6252691b205b0c17b43c55d5a1340fa933ef439b8d4a511:action", "state_id": "5fc8652fda26d19bf6d19cae4550c136a216805e83e8fccd7dc5cc637d85ce94", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.359375, -1.28515625, -0.96875, 0.96875], "student_probs": [0.07239224016666412, 0.0779695138335228, 0.1069888025522232, 0.7426494359970093], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "51d35e2c128104cf52823d8f2a9cf1bc6ee2ea87eedc58b6f7b14a0b7ed94561:action", "state_id": "208a222a7b876ccc4c98405774a6e07a0a1695bc25f490dd8c16af59c983c1ac", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3125, -1.27734375, -0.955078125, 0.98828125], "student_probs": [0.0743638277053833, 0.0770246759057045, 0.10631341487169266, 0.7422980666160583], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a36934bcc88bf8e0e3f9b2b5d60b91678530777ec250977c01f9fc36674f0c9a:action", "state_id": "2846d203fca0241d271433a10ebd5d8b6de0804cff44285cb1c42a941f714404", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.41015625, -1.3515625, -1.0859375, 0.91015625], "student_probs": [0.07340985536575317, 0.07783973217010498, 0.10152214765548706, 0.7472282648086548], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "96662736ea7cdcaf817665bb2d7b2fef6890a0046ee36ca56bace11dbc5284f9:action", "state_id": "144e18c16d3c1f2f66a0bd0e1b7cecb6a2416a905613672733bad39942463f16", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.47265625, -1.32421875, -1.14453125, 0.787109375], "student_probs": [0.07616595178842545, 0.08835405111312866, 0.10574595630168915, 0.7297340631484985], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "294a586abe92f5338a29049410248f11ed13a327d47404ad9bac822bf4d209c1:action", "state_id": "d9773cb0a240d83fca63b38ef0cba5d834fd6ff3c5dac20e75d686c2de84c803", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.44140625, -1.3984375, -1.140625, 0.7734375], "student_probs": [0.07965083420276642, 0.08314792811870575, 0.10760141164064407, 0.7295998930931091], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "78c926e227408bf2c7322a50a8ec64a06753764d3d6d06d5cb0b484b255fc65a:action", "state_id": "1045605c73a743ddfa8ae7a2d6033385e42cfe1f385b0d96d2e216eedecb07ad", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.09765625, -1.40234375, -0.228515625, 0.546875], "student_probs": [0.10751768201589584, 0.07927856594324112, 0.25641465187072754, 0.5567891001701355], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d557a91c0a5c071bae123aee67ea2a374c63e6d40b3ffc748e3f937968fef03e:action", "state_id": "da3b036b7091526f551795b446423a9f7fa373f7a7123bfba4fdc7507a4953c3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.97265625, -1.033203125, -0.03515625, 0.8203125], "student_probs": [0.09521905332803726, 0.08962490409612656, 0.24315036833286285, 0.5720056295394897], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7b83883802b3ec3fbe8fa3302f092104b614df273485e7a00ec878b40f39def8:action", "state_id": "cca1e38276b39a506923e75c889abb543fa340ba0c107c9786aecac143ea5ef8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -1.037109375, -0.04296875, 0.75390625], "student_probs": [0.09364157915115356, 0.0934588685631752, 0.25256332755088806, 0.5603362321853638], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c4f385702b0ef1d54ef41810e4a6edad781eaed44f77962d0d3ec00755228557:action", "state_id": "fb4de4b1fba3b4f13eaf7d1e4f283d4325e5476f2d94d521f012b33c60b9a04a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.021484375, -0.9453125, -0.00390625, 0.78125], "student_probs": [0.09164351969957352, 0.09889692068099976, 0.2535305619239807, 0.5559290051460266], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e980de2a680cef3870c91ea1c6281b302c388fd1edd1f50abc4d130e71935b0c:action", "state_id": "38291ad4988ded63ec3870ac32384b26f9e340d66c7cb8879a35c9c9abd1ca27", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.009765625, -0.916015625, 0.005859375, 0.798828125], "student_probs": [0.0912305936217308, 0.1001972109079361, 0.2518957555294037, 0.5566763877868652], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0f7de4d2ecf66a36bcb624503e4f612ea0b89ab63c85de04a435374cd5e35e37:action", "state_id": "c368c89c254328d6b3cbd0f7297eafb8bc82e3344be8bcf4a1f002d51c4c1c1b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.02734375, -0.919921875, -0.005859375, 0.814453125], "student_probs": [0.08929703384637833, 0.09942366182804108, 0.2480059266090393, 0.5632734298706055], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2c23abceafe98b96a6d4662479e21c0648a80e3c59ff6b6313df2e7a97c9b216:action", "state_id": "aeb04e794d5353a4050ef5fbacef6a53c2608be923528b8545397346c424e7e4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9921875, -0.916015625, 0.015625, 0.82421875], "student_probs": [0.09117043018341064, 0.09838639199733734, 0.24977067112922668, 0.5606724619865417], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "84b97e91801addadaf2e7c004f95854ce858721f7837eede12b1d88b7ff15928:action", "state_id": "d46427d1bc5a66707cc38803d4e5eda0e2b4e9027055d58409b3497fc9e9be1c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0, -0.91796875, -0.009765625, 0.83984375], "student_probs": [0.09031049907207489, 0.09803111851215363, 0.24310369789600372, 0.568554699420929], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a095c94eaa474059d5eeaae125b92d5391e03768c5d7be2c2366438e8f0c9a43:action", "state_id": "ff3c979bc7351a16f856bcd9d5b4d9a24615e7c925e2a2374cd55e332d547803", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.990234375, -0.921875, 0.00390625, 0.81640625], "student_probs": [0.09205470234155655, 0.09856758266687393, 0.24876873195171356, 0.560608983039856], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2a9f23548c8a1a1df28f1b1cba45c5e0c19ced6c070a2b4cc43706280a38cb45:action", "state_id": "9ca511849a6f67cf78afab7a6b57542338e5d5650e40d002c17ed5dc89ed2394", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.005859375, -0.90234375, -0.12109375, 0.849609375], "student_probs": [0.09152334183454514, 0.10150516033172607, 0.22170765697956085, 0.5852638483047485], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0f5dcedd05187a6a94015d72eaa478003b83f98764999de9d5671936cc49cfdc:action", "state_id": "43c2c9474a564eea3cc77800db4f48e52bf2e9e2a97ebb13b49722d4b02b6500", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.880859375, -0.208984375, 0.865234375], "student_probs": [0.09056855738162994, 0.10465177148580551, 0.20489822328090668, 0.5998814702033997], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "13b2a89c2f6e9e36b8ccd4b3a3ca6d641da001b152d8b58f58984d621109f765:action", "state_id": "5f185fe807112ef4cb8a1786a751ae5d3fd197c37110fa2232deb61147ec9674", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -0.931640625, -0.212890625, 0.775390625], "student_probs": [0.09474791586399078, 0.10569894313812256, 0.21688014268875122, 0.5826730132102966], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "abbba171683d0b628b313bba1425a6f8121b49b9b7dbd253233ce6049a22401b:action", "state_id": "5e86f0e74310371711b97e31d0c4b58e32ede58cfae88372991556fac067b0da", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.068359375, -0.98828125, -0.228515625, 0.73828125], "student_probs": [0.09533495455980301, 0.10328318923711777, 0.22079624235630035, 0.5805855393409729], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d55dc8bea11bca05746b7abde0c8390bac81fb7ae38d45a3d55e78f0a321c112:action", "state_id": "b971993998014f560fb066810e459fec9823572e59d0a9030a3d58f02e3a8e0d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.072265625, -0.99609375, -0.146484375, 0.720703125], "student_probs": [0.09424851089715958, 0.10170809924602509, 0.23786810040473938, 0.5661753416061401], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d21a8b018b61db9ab3222768ac14a0db07a6d4c171a046650a5a2c6c0f3999d3:action", "state_id": "106ee89b334a2244ed9f7fd93d7cd4be173edbc8fb510479cd440cd08ad3dc49", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.021484375, -0.955078125, 0.025390625, 0.7734375], "student_probs": [0.09143776446580887, 0.09771594405174255, 0.26048195362091064, 0.5503643155097961], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c8de21bc5fb523af1206af25095606965cc72bcf5ddb610d50a43eb66f4c5ec3:action", "state_id": "8f66ee89c81bea5ab2503e6c24a928e3f3d501aa0150a2c150959111f2be5248", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03125, -0.9609375, 0.017578125, 0.7890625], "student_probs": [0.09008250385522842, 0.09664441645145416, 0.25712284445762634, 0.5561501979827881], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "18883d338b4cb7da06b604cb0ac9e0ed9cdf31b06880465abcf19b42d56a10a4:action", "state_id": "09229234c428a7bf7009201f7a80452c818a5d43673e55a02139c40623b901a2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.037109375, -0.962890625, 0.03125, 0.78125], "student_probs": [0.08969105780124664, 0.09660106897354126, 0.2610548138618469, 0.5526530742645264], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c7a92d394cb00729deafaf81fb6e760367730af4186f408e2e153c48eef4ffb7:action", "state_id": "3b62959f4f386ff07a84f466c1fd75a432acef1320fc4eb6bb5dcc5a9e9c7918", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -0.94921875, 0.03515625, 0.796875], "student_probs": [0.08886842429637909, 0.09684331715106964, 0.2591661512851715, 0.5551221370697021], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4866783917d53b918b5e1e6c1dbbc8cfddc8c1a8d0c97dbe72cad575f545193b:action", "state_id": "cf0e8c61d52135f8bf44230a4d1a6c840703d51f111e3bfccca99bf77337f74a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -0.953125, 0.03125, 0.8125], "student_probs": [0.08759326487779617, 0.09582731872797012, 0.2564471960067749, 0.5601322054862976], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "36c079dd04840a4723e3d7009a725bac4a3ab52557ce6c33455965fa9c53f0fa:action", "state_id": "b3bca80562c2c20cb0f9fdd0f75d7eed945caf3f7559af916b2452027b91cb78", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.037109375, -0.951171875, 0.04296875, 0.814453125], "student_probs": [0.08768536150455475, 0.09555409103631973, 0.2582254707813263, 0.558535099029541], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "951056bd279e3010eb19e8ba177e74086cf95c9fa0d1f361c4af71ad7a4f900e:action", "state_id": "726255cc2946039bd08a2f5003f4e114d24504d17f34052d9b9003516ecb80c6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -0.953125, 0.017578125, 0.8125], "student_probs": [0.0878993570804596, 0.09616218507289886, 0.25384894013404846, 0.5620895624160767], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "00b10f3af63eae621f106fea09647f1953d5d56d0fa956fbdee246e8acd1b605:action", "state_id": "ad53ccf2bbd8131ead148d5f33eb21168bfa4819c88490a3d4be1d9cb6996f17", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0625, -0.93359375, 0.01171875, 0.8046875], "student_probs": [0.08668995648622513, 0.0986170545220375, 0.2538025975227356, 0.5608903765678406], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "69bd78381fc9356534c1fa76aa7c1c09dc930d9ca4bb2e3261d06ed51934ad97:action", "state_id": "a557a4ae5ac67dbbf40bc4fb94b06e201a3527adc2fe8adc31e48b3d33f58609", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0625, -0.96875, 0.025390625, 0.810546875], "student_probs": [0.08639766275882721, 0.09488926827907562, 0.2564288377761841, 0.5622842311859131], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "03192be7599cb6eaf79a74cde9441dfb3874b0f6fa5f14c9be8a2bb8a7fe8226:action", "state_id": "2fdab9bfb3e5bd60f51ca60123f485e5c924b515bb39112f27ec97b32b3a3e7e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -0.96484375, 0.03125, 0.8203125], "student_probs": [0.08746293932199478, 0.09438546001911163, 0.2555660307407379, 0.5625855922698975], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dc7cee430c888bd7524954bd960b7b8907a1bd7fc8f2a05cfd9af3ea04af609d:action", "state_id": "b1e099222462ed60c6e17074f26e21520804f48964ef8a53823c1019be261fce", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.060546875, -0.998046875, 0.03125, 0.802734375], "student_probs": [0.0870400220155716, 0.09265362471342087, 0.25934645533561707, 0.5609598159790039], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6b2e2473bbceb6ac1028f2b82d30842b5f2dffd4a762428272ed4375257e43d9:action", "state_id": "dd85f8fb94dc2d5f6d2afc74542834294139be9084c7be359a3158a07c800a9f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03125, -0.986328125, 0.064453125, 0.837890625], "student_probs": [0.08680589497089386, 0.09079429507255554, 0.2596611976623535, 0.5627385973930359], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e24dfd9832cff4c653c6b877f3082635a338db3d3213f4e3f5a9a06dc57c3cb5:action", "state_id": "059b7756b6e57e8ee04b3c6c93a0647b0bcc378bb01677ad436a3143d3b494a4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.046875, -0.998046875, 0.048828125, 0.8203125], "student_probs": [0.08687042444944382, 0.09121740609407425, 0.25985419750213623, 0.5620580315589905], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "651f52ca1c376f2f1418d3d7b1049e0984ec5f3fff48eceb0d8f082424a0b92b:action", "state_id": "538643b3f248fefd4576d2cfa9c22e7d40a5c6503c755f6a413388a5548f3379", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.044921875, -0.994140625, 0.01171875, 0.8046875], "student_probs": [0.0886044129729271, 0.0932200625538826, 0.25488752126693726, 0.5632880330085754], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a5ed76ee67632b2a839eb5d122d17264f268f0ddb67dc0c21f2fe534afd875d5:action", "state_id": "b36b2a39c4eededda91591a0eccf68fa039a87d4ea011cf918cd9b2b0251a8af", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.046875, -0.994140625, 0.03515625, 0.8203125], "student_probs": [0.0871468037366867, 0.09186577051877975, 0.2571411728858948, 0.563846230506897], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b23d85d0f5af2b43255657cc7f03a7a2ad6531f6aa78b932247accc615f98309:action", "state_id": "a8ec359910abba03b076743ae8a83cbcf81f19922ecad0147a4e2a50cc0e6d7d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01953125, -0.9921875, 0.060546875, 0.828125], "student_probs": [0.08835825324058533, 0.09080763906240463, 0.26020708680152893, 0.5606270432472229], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e5a9c975b43be62a47846e4a85f494ef7edd02afdb3c365edebe0019975ad208:action", "state_id": "171ccd68643ac4a2e77caf2b55c7b11f806e1d3cc00061fd5586a2809c4aff19", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -1.017578125, 0.021484375, 0.802734375], "student_probs": [0.089146189391613, 0.09108216315507889, 0.2574497163295746, 0.5623219013214111], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "74e5935f62b7c30723ef3910d1ca32e33be4f82d67c7e3831b5e0aa0646db3a0:action", "state_id": "30a6f4f862f6fcd48e7ae6979a5ae12fa84a8a89bf1c4ec46f919190bda12e35", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -1.037109375, 0.04296875, 0.802734375], "student_probs": [0.0891227275133133, 0.0889488235116005, 0.26194626092910767, 0.559982180595398], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "11160a34a66c26d5f7324a9ed1b7b08f5156ff8343a96c185dfd6559377f9d46:action", "state_id": "6048e61f9822bd7e78802d1be93a3f279b479a576606c507b1295c91bf0686ee", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.017578125, -1.015625, 0.05078125, 0.810546875], "student_probs": [0.08980881422758102, 0.08998439460992813, 0.2613975703716278, 0.558809220790863], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1b7ae96ffa6020e381e447570cc1f206c81b6a6aca2f02aed7f1edc738298b68:action", "state_id": "c9c9b5563e67cbdb1295f2f15b2118f379b763104ad30aff75c771d65f289910", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01953125, -1.0078125, 0.029296875, 0.802734375], "student_probs": [0.09048177301883698, 0.09154834598302841, 0.2582625150680542, 0.5597073435783386], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7d19ba26833ee4add46834f3e6e31e71e9a7d88202fa0479df58d11a0080d9f7:action", "state_id": "d4e56524100330411166edac9fcabd3079570d9550e18bb6a27a3d21353b58e7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.048828125, -1.013671875, -0.029296875, 0.771484375], "student_probs": [0.09106253832578659, 0.09432089328765869, 0.25241580605506897, 0.5622007250785828], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cb18445bb33fae0f5ca2c30d6a09b145815abc7b31edb6a1db85e50f54955a6c:action", "state_id": "dd429998ed6f1ed5d3d4c24f8b0f036e55d9bcf3adddabb89528ba67394b914a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.056640625, -1.080078125, -0.1171875, 0.703125], "student_probs": [0.0966511219739914, 0.09441220015287399, 0.24728979170322418, 0.5616469383239746], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1b7c4b9b4f75a87c81cf77d5404fba660ea824e09c5cf2e201de400a9b496197:action", "state_id": "7c6b9bace3f2d5e7f61cb3f6e0a1af5ab88243fb7c212ab8954347d9a41b9460", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.048828125, -0.873046875, -0.076171875, 0.767578125], "student_probs": [0.09101784229278564, 0.10850940644741058, 0.240738645195961, 0.5597341060638428], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cffb925f11259352f9d6f117b51329a01512c43ba03a7d220bcd0ca0b66154c7:action", "state_id": "a63de21bfec52c67a13bb1e847d70cf295cb28ab068ce5844f77a128f72e4ac4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.65625, -1.77734375, -1.57421875, 0.451171875], "student_probs": [0.08929812163114548, 0.07911375164985657, 0.09693218767642975, 0.7346559762954712], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dc9ffa732367a40551d96582440b370b762402a84a2ff8ade31a176bf7db770b:action", "state_id": "dacb348308500b5d6b1ec6e188f1a5bda98632cbcfc24786b4befd41d07125a4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55859375, -1.60546875, -1.3125, 0.47265625], "student_probs": [0.0921078622341156, 0.08788993954658508, 0.11780776083469391, 0.7021944522857666], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "abaa46858a2f145400bdd134f218a07f605c070d000cfc16aaa2fff829ed66d4:action", "state_id": "2eadc688c25b86f83721f443f91515dfe473165f60ca8d69319f5bcb9c342863", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.57421875, -1.5625, -1.3125, 0.474609375], "student_probs": [0.09033625572919846, 0.091401107609272, 0.11736135184764862, 0.7009012699127197], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1618d8d414c45b05d28f7b3aa16ca66d58200d4f5644b6ea80d2ba93b6e6b5b6:action", "state_id": "0efb9d820690cb73c0346e3420f32f478c9ac7bdad34c1a6baea109e8b24d8cc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.53125, -1.359375, -1.2265625, 0.478515625], "student_probs": [0.09086532145738602, 0.10790524631738663, 0.1232316642999649, 0.6779978275299072], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "815755b71c9cd3f780da7e7478592aa4336fbaa0a808ae0630e9fcf8c1202ec4:action", "state_id": "065dc4ec8131fcbf0137baa6fa8dfe2407771a46ea21af3f067f258058118b3a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.095703125, -1.43359375, -0.275390625, 0.474609375], "student_probs": [0.11373157799243927, 0.08112169057130814, 0.2583082318305969, 0.5468385219573975], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0b95fb42296c7f953906bd3393847b44a836f4c639abd071efb89da186b5fde4:action", "state_id": "e74d34f133d5621a3aeefc86a11e44063345c7eb2d9128960b73f538956e176e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.96875, -1.080078125, -0.11328125, 0.75], "student_probs": [0.10178454220294952, 0.0910610482096672, 0.2394457757472992, 0.5677086114883423], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7bc528177f590737653b2463275c8561c75e6e12fa96eeaa7a61c6314759139c:action", "state_id": "2dd8979407a1caaf050d227729d75ff94075103e1cb5785c2057ad2810aff6eb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -1.1796875, -0.1171875, 0.658203125], "student_probs": [0.10214338451623917, 0.08822525292634964, 0.2552882134914398, 0.5543431043624878], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2d8f87d1967ae955c2cbb8d285ce3293bca577fb30f744216fa529de4bfcb477:action", "state_id": "78b55f7732fe9a5e39bb4cfb5a6e9b5394aca55dd06f7038533a917284b6ff45", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.810546875, 0.001953125, 0.78515625], "student_probs": [0.0890735387802124, 0.1112876832485199, 0.25079065561294556, 0.5488480925559998], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "328e634349bf5943ff75dcd4e30cd381457130f2f7b0e446832abb2174dce8df:action", "state_id": "2edae292e4a981c212f16ef65e630f0cabda8a47bf4059420a3193915698b15a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8935546875, -0.796875, 0.10546875, 0.841796875], "student_probs": [0.09534654021263123, 0.10502492636442184, 0.25892579555511475, 0.5407027006149292], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81c85b13e66d666927303c93492e223bfd5518a40702fc2e99cd1c70867db90a:action", "state_id": "997ec9036c4548638f65a6d73f81747c594f91a20879d240ea886c4eab64df30", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.96484375, -0.8828125, 0.072265625, 0.826171875], "student_probs": [0.09172562509775162, 0.0995672270655632, 0.2587626278400421, 0.5499445199966431], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ea53823921e9da06317541b1d8055fad56a72dd43237bba32ba3abd2ac53eb27:action", "state_id": "bc9052b29db7aa0ba6a0b438cf5d6239c1bff02091ba8f2d08fc7d47d0e20a23", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.951171875, -0.873046875, 0.09765625, 0.8359375], "student_probs": [0.09167813509702682, 0.09912770241498947, 0.2616772949695587, 0.5475168824195862], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2b7666b32a508059881e5154defabb98d8956c06e33f7af92894a9388a850a0b:action", "state_id": "d62585133f927d4459aeb1c02ae862f5e7f634dab5d760e18181be0698a612c8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.94921875, -0.873046875, 0.115234375, 0.861328125], "student_probs": [0.0901535153388977, 0.09728898853063583, 0.26137784123420715, 0.5511796474456787], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d06797cff653aa284960a4ecf8871359fedd66de0990d35f4a1481b1b6918dd7:action", "state_id": "0e59f6fb5ed8156e0da97d9b87e15356a3c7963f932efbfdab76c4e7de690d33", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.8505859375, -0.837890625, 0.134765625, 0.90625], "student_probs": [0.09536884725093842, 0.09658729285001755, 0.2554696202278137, 0.5525742769241333], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "abc2fbb86cec878de503640617557d8d0edf314c179f335491e270c0d387985d:action", "state_id": "e6e9de52f6046120b591cfb86761dcbd9c45305fff9c46e1633d307e9ff16cde", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.990234375, -0.912109375, -0.0078125, 0.81640625], "student_probs": [0.09223280847072601, 0.09972743690013885, 0.24634617567062378, 0.5616936087608337], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "66b2622422ce69617086c4829a9d0403bc66c999ac154d30361d4c1109d6a4d0:action", "state_id": "d5f2b30dd1f0938b93d5c18b59ebd9e113aeefc0d49501a57a80e6c5af8e63ec", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -0.931640625, -0.03125, 0.748046875], "student_probs": [0.09237730503082275, 0.10285324603319168, 0.2530770003795624, 0.5516924262046814], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cc95d1b9fd3e446264fd5a55a239e65efdf5db4db56024fce69e15c883980ec8:action", "state_id": "1493d9cb057cb2020accf4cbf477b21bc134270a6ba5c830ad1960599026d801", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -0.955078125, -0.0234375, 0.765625], "student_probs": [0.09233351796865463, 0.09944711625576019, 0.2524634897708893, 0.5557558536529541], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bba509b100f3a9f6c3b1451037d33f76db36b2a57fdad9497a7e20ad88b523dc:action", "state_id": "06b4c3265ebf452210b29d07b7e8c7b08eff175bd1780eca9add825cc5b5978c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.943359375, 0.00390625, 0.78125], "student_probs": [0.09111998230218887, 0.09890980273485184, 0.25505366921424866, 0.5549165606498718], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f33cda7bd5e2fb8e6eef702eba64eb22059c3adc38d44cb2db0a9213a57b286f:action", "state_id": "55edd5a57afada479dbe62d7a918a12d17e544ba2f351a6b5f8851d6dc784792", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.955078125, 0.001953125, 0.78125], "student_probs": [0.09062467515468597, 0.09798863530158997, 0.2551579475402832, 0.556228756904602], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5ee78206686a0f50df0d4b0f9060605388ec478199d290bfeccf6ac70a82d7f0:action", "state_id": "5d4d9b46660bd9e4fee5cdd7adf58e2f102af4a11848da8879d95e5d91980a6a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -0.943359375, 0.025390625, 0.806640625], "student_probs": [0.09014783799648285, 0.09690359234809875, 0.2553069591522217, 0.5576416850090027], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "34c131fb3dab161ceaf7152ea935a097f87f711d42f769a6fd66c686526fa966:action", "state_id": "7af91864fcdf490fc99a7c4fdf13ea25cf559720084ac20ff40a2bfe510bcc2c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -0.951171875, 0.025390625, 0.796875], "student_probs": [0.08910968899726868, 0.09691675007343292, 0.2573443055152893, 0.5566291809082031], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "150c16d23833402eef8cfb0ecce8d3c21372daf4a52f76782d95123d42ce9700:action", "state_id": "281f4d0754f108e5b4a3a008df1f93fcce9a5f83d7e2b39db4a950db9036adf0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -0.958984375, 0.025390625, 0.7734375], "student_probs": [0.09034273773431778, 0.09749318659305573, 0.26090529561042786, 0.5512588024139404], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fa95356e4a3b515e2d1845dca010bbccfe184f8d7358d24e8bba7525e0513c5b:action", "state_id": "74584c9fefec0db7ab08e650e104ed541e8608b6361e77fad2525551ae2faaae", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -0.94921875, 0.03125, 0.7890625], "student_probs": [0.08982254564762115, 0.09731120616197586, 0.2594030201435089, 0.5534632802009583], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "303bf68b8ab79fa05aa07f5696a2805ef2479f98df25cdc1880c9a98864f1cbc:action", "state_id": "839859e6bb0f6d2172f07769f2c84cd59fc2fa2f211a0ca7930801805f614356", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.037109375, -0.947265625, 0.0078125, 0.7890625], "student_probs": [0.08970825374126434, 0.09814111888408661, 0.25505635142326355, 0.5570943355560303], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d3f74e50c75b424861506411d0cc7021d06a29a5b1ed45fa5786f2c5c62e995a:action", "state_id": "e31219f1e10f3a26c2fb783eb26c343355164cef4698d0c75667b59bba5f1d0e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -0.94921875, 0.0, 0.796875], "student_probs": [0.08967125415802002, 0.09771818667650223, 0.2524735629558563, 0.5601370334625244], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "64122c26a1fc38faaa674dfc25b771e2428b17d8103b2926cae15e6904d39231:action", "state_id": "fcf7d70f9f35b879052f1ee3fd662042c607ae2cdff8b10c538a6482f370eda0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -0.953125, 0.021484375, 0.7890625], "student_probs": [0.08928652852773666, 0.09729894250631332, 0.25785502791404724, 0.5555594563484192], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8e231910a171230b8880499123e5717ed0903e52a713ae31feb3936e09e6a61c:action", "state_id": "76d41bf37bd21ff3d67cf6ff0f496e2eec15554ad8322988a7a8be0550d045b8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.046875, -0.927734375, 0.013671875, 0.78125], "student_probs": [0.08899419754743576, 0.10025449097156525, 0.2570107877254486, 0.5537405014038086], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "74c68d6bc6d1a061bcc2236ea420509271aa27ab35bc34691ec0ccdb5c8dc4c8:action", "state_id": "3a6f65c9f9d7f2c6dcf93436a5d1196f0fd28b642216fefa3c0ceaddced7c3c5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0546875, -0.9609375, 0.0078125, 0.787109375], "student_probs": [0.0884975790977478, 0.0971955806016922, 0.25607624650001526, 0.5582305788993835], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cb345ffb853d1662a8220f6e2a3489f547be0d7f243fe5d6541b626efa19a6c1:action", "state_id": "6afc14a1265b74fa2ac300a59bed7371dbad4a46917b249e17608cd18a88f041", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -0.958984375, 0.017578125, 0.796875], "student_probs": [0.08935604244470596, 0.09642839431762695, 0.2560475468635559, 0.5581680536270142], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "291cdeec426190713580087de880000c99df9702a99a51893d7b1be4bdcb8d4a:action", "state_id": "638bf5d34772585d3250980d0a9b2e602ff3abc96c6be14fa6de51f3625cc717", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -0.984375, 0.0390625, 0.796875], "student_probs": [0.08955265581607819, 0.09366725385189056, 0.26065200567245483, 0.5561280846595764], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "82a0308b7882028381394a72dbbb8635765dc50b7f5f650a2a6a92108c8d139a:action", "state_id": "ea20b249826488ab658c31f5dfabd86a14ce0b61ba1311b017999e114de37c0c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0234375, -0.966796875, 0.060546875, 0.814453125], "student_probs": [0.0885113850235939, 0.09366942197084427, 0.26167821884155273, 0.5561409592628479], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "86b3205ddbb338153d538e2ca76a4f9d1ec1f3357ad46a1ec676e37a9be58969:action", "state_id": "aa42652e6c74617b052f68f18b6d5a864b0b7ae2176faeac00d0d7b487e35b02", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -0.984375, 0.03515625, 0.8125], "student_probs": [0.08807717263698578, 0.09302803128957748, 0.25786396861076355, 0.5610308647155762], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3f3e36c4975433d875b53437efa38add43f527bbf1bff063682080f0d782f016:action", "state_id": "50058836d36cd710f1e8d34621d97d31f8791442660e7fe3f0ecc1be47d56ae5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -0.9765625, 0.03125, 0.830078125], "student_probs": [0.08801404386758804, 0.0927799716591835, 0.2541801631450653, 0.565025806427002], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b9f12a601975b9c936aa43fa8ed453e4252a2a02f678b7254cb8837562826e7f:action", "state_id": "0491eb940894e591585b42407ae4292ee4deabf1d6ab5c41ad26bfbace092a8d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.021484375, -0.96484375, 0.025390625, 0.822265625], "student_probs": [0.08906935900449753, 0.09425991773605347, 0.2537350058555603, 0.5629357099533081], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "daf01ec62a42f72acbecf9b018a4ab017940edf4f7ccf6039ee23becc8705db0:action", "state_id": "53b5ca522d097c6225b892147c0936b86afab43d2124e97032d46183f6344def", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01953125, -0.998046875, 0.05078125, 0.828125], "student_probs": [0.08862939476966858, 0.0905541479587555, 0.2584690749645233, 0.562347412109375], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0bae7add0ce0fac22667663aa12b023348c0b57d29da5b1492c85f9bfec772cb:action", "state_id": "4482a61bb06ff71f397916408e8366a6b844d0905261a2f71eb02ae8daf88de6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.021484375, -1.0234375, 0.03515625, 0.802734375], "student_probs": [0.09031228721141815, 0.09013606607913971, 0.259800523519516, 0.5597510933876038], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a7bba5ccb0a676f63229c32caa3e979a4513f0acada2b7bc5d3d9ca00d07d288:action", "state_id": "56b6ea37257e0305d0e0544a9abe1054351443212dbb4aae9b4a6242492cb975", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -1.07421875, -0.0234375, 0.751953125], "student_probs": [0.09442811459302902, 0.08992812037467957, 0.2571840286254883, 0.5584597587585449], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9d7314aea870aebf1acf550172d1ab95ede2e55062ec6bad1744ff56c11f0884:action", "state_id": "f9421b03b4ca8f44c757ec3c0ebc79ba7dce758358143e87782f2051605c8026", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.00390625, -0.8388671875, -0.05078125, 0.77734375], "student_probs": [0.09336619079113007, 0.11011974513530731, 0.2421734780073166, 0.5543406009674072], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "783339fa100f39501d386a109349088186dee12e9070da43a4c21b52597962a5:action", "state_id": "7d7a8eb530f9df6b87bd6532c3d94886391898ee48e939a8ab10820f07705571", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.021484375, -1.40234375, -0.216796875, 0.91015625], "student_probs": [0.09242064505815506, 0.0631486177444458, 0.20665234327316284, 0.6377784609794617], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dbd1daee1150599de4827eae3cffded2aa1b970a83137d7eb444d5a0c35dbbeb:action", "state_id": "61649df0615c2ffa7133feba986c1b8e8a3b24f2b8400d073176afa18faffb0f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.970703125, -1.22265625, -0.05859375, 1.01171875], "student_probs": [0.08675166964530945, 0.06743044406175613, 0.21597424149513245, 0.6298436522483826], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0d9eb7aa60bd219017cc9f05e53419e502d158e9b68bff29730bd014c2cb228:action", "state_id": "14164ed42ea11e1b2cbdf73ca53506b9fbc23cc70030a627eeaaeccd6c94464f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.12890625, -1.23046875, -0.259765625, 0.615234375], "student_probs": [0.09990741312503815, 0.09025882929563522, 0.2382652461528778, 0.5715686082839966], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fda80fd58b9a8372cbdc943757531c0e8a211cfc254695c3d36fb5edcc5706c2:action", "state_id": "8775db838b971e4c2a0e5091c04732ceba1d5f2c38a1fadbd2d88acc4d84c055", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.111328125, -0.888671875, -0.115234375, 0.73828125], "student_probs": [0.08838285505771637, 0.11042474955320358, 0.23931288719177246, 0.5618795156478882], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2cc3b58fe12ab0447a1e6f0120a2e9c93a68b2bb147bf26632914539916f4b95:action", "state_id": "31a7b295a1fab14936bd0e520e23a3bee87617da455099d673a37958edf0f8bb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.982421875, -0.9375, -0.068359375, 0.759765625], "student_probs": [0.09755905717611313, 0.10204152017831802, 0.24335481226444244, 0.5570446848869324], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3bcb6fbd8fb5c62b8d2f3fc19e8688562defe9ca2b249ee9a83dd6122832c205:action", "state_id": "54ce4033d4ef2e3acb4dae7fb9ed0c5c5483990133a07d79ff410b025b782ef2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.044921875, -1.390625, -0.2421875, 0.91796875], "student_probs": [0.09042216837406158, 0.06399381160736084, 0.2017892450094223, 0.6437948346138], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d04ebe79c88a35b75a32a5efed25ef8266159508209fb76b77c5485a4d60caba:action", "state_id": "cf660c1eef5ac54d05c182597c7a1a020ae8d563acfc1f0b73358db9a071f876", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.30859375, -1.23046875, -0.3408203125, 0.91015625], "student_probs": [0.0718950480222702, 0.07773707807064056, 0.18923333287239075, 0.6611345410346985], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "156f9589826168dfcb8166609a3a9a5ab102477e74c74eb3a960950686b6c0da:action", "state_id": "f271f79942e03fddd58484c4e86d476b5e6789ac1f74985bd4ad2a3a24df4662", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.33984375, -1.328125, -1.1328125, 0.81640625], "student_probs": [0.08417161554098129, 0.08516380190849304, 0.10353285819292068, 0.7271317839622498], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "89970f2659b454a8709ac6ac53becfec177b8093706d017d4f93ce15d74901dd:action", "state_id": "92578136d3f08f7fd7d299e2c84f846ee3be08137bf763054279ad4a628bb07e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.203125, -1.080078125, -0.87109375, 0.9375], "student_probs": [0.08312907814979553, 0.09401378780603409, 0.11586500704288483, 0.7069921493530273], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6751878e8f2e52630ce622499e7ce04f51b5341578e714180ef7db4f5e92e493:action", "state_id": "590b8691780c2638ac5c18c4b6e5f993326695ecb1a5cebfea8945c8d63a1912", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58984375, -1.46875, -1.29296875, 0.7578125], "student_probs": [0.07175968587398529, 0.08099736273288727, 0.09656321257352829, 0.7506797313690186], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d97a68bfd00181c6cfab3e58a895ef75d82b194ace14fd1f287cc3951c8dd2fe:action", "state_id": "8673264df8876e6e7ce97297ce1cd49f887c672024b6a967172dabda8cd05b19", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5, -1.49609375, -1.16015625, 0.6640625], "student_probs": [0.08254177123308182, 0.08286482840776443, 0.11594874411821365, 0.7186446189880371], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a73447fdcb6e3208b3a07fe753eea95cc0b2166a212c755edb87e6808e755a3e:action", "state_id": "e4bb6177c019d3fe2a6aa60fb03774564e35db360d9e467354e6b92dfd5dec8b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.56640625, -1.6171875, -1.3046875, 0.890625], "student_probs": [0.06702519208192825, 0.06370654702186584, 0.08707652240991592, 0.7821916937828064], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a3594287e0661d3bd415b9ab7e5c21056d35ffc7329cbd710d3662b76483ba57:action", "state_id": "271dd5e3d21bc044d67eea0c0cc2a4ce918d64bb86e235e0aae3ba3d5870781b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.36328125, -1.3203125, -1.064453125, 0.91015625], "student_probs": [0.0763072520494461, 0.07965753972530365, 0.10288337618112564, 0.7411518096923828], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7aa95afdae22878a7050376ddf96e3f382ceffcbcbabeb9bb23ba04d0f34a6fe:action", "state_id": "8520c013e7a453f2980a23cd91c59a22689e2564ee8e3fa6bccfa742754b7425", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.33203125, -1.3828125, -0.9609375, 0.892578125], "student_probs": [0.07905340939760208, 0.07513920962810516, 0.11457361280918121, 0.7312337160110474], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8f4e53986e2899b73c85d7d5607578ee307b4ba64ab3605415588865ea449c7c:action", "state_id": "5ee4bcd7900ca02f1cb99411fb057d4701a908197cc7ba377646c8571f5fbc7a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.29296875, -1.2734375, -0.87890625, 0.908203125], "student_probs": [0.07956544309854507, 0.08113472908735275, 0.12037866562604904, 0.7189211845397949], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3580f8c9418205e42a2f8a4dd77988edffdb1bbe034386a6dc12fcc8f6e1a447:action", "state_id": "0e21cbb0ca1ddf82d5b9fc76cced01a82d18ada6eb7ab6dadfe315e86562024f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.080078125, -1.083984375, -0.611083984375, 1.013671875], "student_probs": [0.08539871871471405, 0.08506578207015991, 0.13650009036064148, 0.6930354237556458], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3d29c8e8d7fc2bb9961ab54d3cbaba1dacfea9bde3d3e5a035a8ee385d601c87:action", "state_id": "1e87dd3cff28fe158eb8bf5b87caeba81cbb3ad75c9f245adc4d5f402f58bd94", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.05078125, -1.037109375, -0.55133056640625, 0.849609375], "student_probs": [0.09661753475666046, 0.09794754534959793, 0.15920789539813995, 0.6462270021438599], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "41ca044a9f9a29447957bca56e09429214a5c119c5788b805854a3a18d163644:action", "state_id": "ea5c23f51064d755aa73f34fde9aa7c81e265056c3e66a0e563f51706ade46fa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -1.017578125, -0.5146484375, 0.869140625], "student_probs": [0.0955052375793457, 0.0977700725197792, 0.1616685539484024, 0.6450561285018921], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "45b87cf13d690d3994935bb786590148b134d7c351fa1cae338e3b837e9d2330:action", "state_id": "e46ac5a9487b7ab52117bfa84233656ffbb02bb12b53f2737208961af9188713", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -0.98046875, -0.44189453125, 0.88671875], "student_probs": [0.09279941767454147, 0.09878447651863098, 0.16927331686019897, 0.6391428112983704], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "35ccab39a754cf0d903e5e432bfb4871bce3d16c2e3437abb0ae1c308ff3190d:action", "state_id": "10504eaa5d928bcb393add103250e9f53bf6eef778349971bd8cd81f24228605", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.080078125, -0.966796875, -0.3408203125, 0.8359375], "student_probs": [0.09084168076515198, 0.1017378643155098, 0.19025705754756927, 0.6171634197235107], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "355a9ce773c614be5bb6e01e8e254899fc1fa9b122ddd804f78183043fe8543c:action", "state_id": "b097ad4c2dd03165d917e266b76a2a47e234121192c789ced6968137892226aa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0859375, -0.97265625, -0.357421875, 0.802734375], "student_probs": [0.09257088601589203, 0.10367447882890701, 0.1918071210384369, 0.6119475364685059], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "15f1773e924e57eba8b0b7199b5f716cb951d71fabffc3d75112ad2444e17bac:action", "state_id": "454120b2b041c3ca2943f7154d80246ec65bbccf916dd9508b41fa2c453b998a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.12109375, -0.998046875, -0.36328125, 0.783203125], "student_probs": [0.09108110517263412, 0.10300703346729279, 0.19433100521564484, 0.6115809082984924], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "569ab5b2f1a23910a95f903ac80bbf82ab061af8b1f74aa2aad9b0607a1eaf0f:action", "state_id": "bccaa920c4cc60d63e3763897c8c9bae12cefd9ae88fb62094e2d20ca1e2e9c6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.09765625, -1.029296875, -0.2421875, 0.759765625], "student_probs": [0.09233248978853226, 0.09886500984430313, 0.21721003949642181, 0.5915924310684204], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4089d13222d13998a72d7ff6c2e44a548b92076c3d72c1e6e7b2d295a51beb51:action", "state_id": "8814a405e7d110c4c5e65e0c532ce0742ed9564ef2f6ba92821d68a6522d55ad", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.984375, -0.07421875, 0.796875], "student_probs": [0.09245163202285767, 0.0963224321603775, 0.239333376288414, 0.5718926191329956], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "da3fd1cf99bc7a2f7a1609a67592329deebb20ab4ef94da3913e5a2f49723d56:action", "state_id": "f52b550876430f0ebb172e532d3177d2c523f8eb370908b7344b3ef18e46e5e9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.052734375, -0.9765625, -0.103515625, 0.8125], "student_probs": [0.08992573618888855, 0.09704317897558212, 0.2323402613401413, 0.580690860748291], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "39dc410c3d4b6d9d365d37cdf996b392aecdebe1ded967028b591ece30798485:action", "state_id": "2c70ac8f9745d3575f785ad5fdeaba0a51eac73f136d99ca95dc2c5ef96e5ba9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.97265625, -0.1015625, 0.822265625], "student_probs": [0.09094394743442535, 0.09662044048309326, 0.23087674379348755, 0.5815588235855103], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6c4c366acf11275d340811ebeef9e3ebd5add8cd90b1fa7d94f97cd995d79029:action", "state_id": "1a06f3c94e5017f11e36f77df61d7675451b1cb7cb8883cdaf1f3fba480afaab", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.05078125, -1.00390625, -0.16015625, 0.779296875], "student_probs": [0.09329213947057724, 0.09776932001113892, 0.22732049226760864, 0.581618070602417], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0ec7071476faff74b56f3389e04c2674e1bf588945bc9d4915b3f01db2b8226f:action", "state_id": "47531dbc50bea3c26856e5d9b9ac5f5c42ca63e2fa999ae8442bcbb028be5c0b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.056640625, -1.01953125, -0.16015625, 0.763671875], "student_probs": [0.09378606081008911, 0.09733178466558456, 0.22986693680286407, 0.5790151953697205], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "89595e88e710b0b59acaa86fb140de9492ef2f5273d092fbe38c5f5cc2ce83cf:action", "state_id": "82ef880105c12df1be7075069e80fecd0209b5b43fded6739ada56669173a605", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.056640625, -0.9765625, -0.2578125, 0.869140625], "student_probs": [0.08955106884241104, 0.09701710194349289, 0.199066162109375, 0.6143656969070435], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "692d9bf386a4ca0397fdd714cc459e057ea113250cd42f8df4ffd93d5820a6fd:action", "state_id": "18f0610656f4a6a8802cdcb4d612e46d5798dd528bc769b9b4a71be39edb1ad2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0703125, -0.96875, -0.3212890625, 0.861328125], "student_probs": [0.08990659564733505, 0.09951752424240112, 0.19014647603034973, 0.6204294562339783], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "48526f76cd059550ff8a1a92d1a6f27b04f60640683a4a7aeb93b38f5c136541:action", "state_id": "5b85b27b5f0c402c6190af6e8b384f424df235869e84306f7b64b709ec639ffc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.06640625, -0.970703125, -0.3583984375, 0.869140625], "student_probs": [0.09043058007955551, 0.09951271861791611, 0.18356892466545105, 0.6264877915382385], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "071abba709b0ae5a5083ea015754feb8cb00b121b26183430e432dd977f7a3f2:action", "state_id": "fab90b6a3abc310ef3f441cb1a5b8aa5c771e0ba7153c3da268e440a92bdd1b4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.09765625, -0.974609375, -0.400390625, 0.859375], "student_probs": [0.08914664387702942, 0.10081926733255386, 0.17902907729148865, 0.631005048751831], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "37466b9efe715064da68d9f4cb82e37917310e0a6ed537fe22a55242d9b52bf2:action", "state_id": "70e9d03204dd4d5bc644c6e225621251809575aa284f5e0b61f8c703f238a7c7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.134765625, -0.970703125, -0.439453125, 0.857421875], "student_probs": [0.08684945851564407, 0.102333664894104, 0.1740754246711731, 0.6367414593696594], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8ec9fd7f92e91dae78990c6f4ac36fa1d775f41401aaa8471eda8ff978644b5f:action", "state_id": "762dca64ccaeaa79137436b8a9fba0515bbc18dc4fe89d4b265fb797891389d7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.15234375, -0.974609375, -0.494140625, 0.837890625], "student_probs": [0.08738909661769867, 0.10438697785139084, 0.16877619922161102, 0.6394477486610413], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9b1854219ff198996a193e0b23bd3aae303c0ebf7e40371824611410f98735ee:action", "state_id": "19923ff6a06c5c643ba89a6aad5564b8a117c99be67bad83526952812527dc12", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.134765625, -1.0078125, -0.42138671875, 0.822265625], "student_probs": [0.08885318785905838, 0.1008806824684143, 0.18133828043937683, 0.6289278864860535], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d90b96eb86a9074eca7f1f561894e82b9487dcdc2501565b623527dc311d1829:action", "state_id": "a7065b224d3ac968550f1028bced24548b884bab215f0dfb3e92e4eba38bb8fc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58984375, -1.39453125, -1.10546875, 0.51953125], "student_probs": [0.08276817947626114, 0.10062050819396973, 0.13434600830078125, 0.6822653412818909], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d81542bb4e0cae7db816ae645a9830bd2788276a6340cedaac1eb965553ba004:action", "state_id": "7f8b4af4c0a21c2732516e8b6c010f780b67ec9ecc471047cd1405db57e998a6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5859375, -1.4140625, -1.068359375, 0.484375], "student_probs": [0.08479759097099304, 0.10069963335990906, 0.14228686690330505, 0.6722159385681152], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3b7a388a9535bfb41894bae188542398da1033409902b69211aa8e0ffc0e60ab:action", "state_id": "3ef5cee109a645d2b5b2be747d8d46aa86c6b0539d7a5b8e9e89be7e1ca88f9a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6328125, -1.40234375, -1.12890625, 0.416015625], "student_probs": [0.08566655218601227, 0.10787047445774078, 0.14179307222366333, 0.664669930934906], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a1d6f1398336482ba16f4fe7d710f2c70d3cfc90518cfae7cbbc9560736391b1:action", "state_id": "c349ba9075836f724cdf39fe754c2f55b16060e126a42b30852c2dafbbcf3d75", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.53125, -1.48828125, -1.1953125, 0.7109375], "student_probs": [0.07777828723192215, 0.08119316399097443, 0.10883139073848724, 0.7321971654891968], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f42073fbb2ecd31f70b6108e37b7a0b369701ef1394df526dd9e8a16bd23e73d:action", "state_id": "50a3d9e989378c3594a5bca1ad967656e9b126e5758021d262423eea4a77d0e8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -1.390625, -0.201171875, 0.92578125], "student_probs": [0.08937674015760422, 0.06313051283359528, 0.20740167796611786, 0.640091061592102], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c6c7d9109f53ef2c42708166dacefffca0a9654971c4f54d5e7d0da1deade600:action", "state_id": "922d1f488f9f8be4b4b26b43029c7c1ca938585c559b868435a805692cb5a1d5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.421875, -1.3203125, -0.475830078125, 0.83984375], "student_probs": [0.07001832872629166, 0.07750321924686432, 0.18033240735530853, 0.6721460819244385], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b226e7e2211eb8fb7084a89896b05d25c4ae2768afe2ad06be6594f70ecb21f9:action", "state_id": "3327a4f420126213c1964f03c12a3aa7f16ab29b394c554fef6777d86b05f20f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.34765625, -1.34765625, -0.939453125, 0.80859375], "student_probs": [0.0823533684015274, 0.0823533684015274, 0.12386874109506607, 0.7114245295524597], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "86c3b68f8c489341cff1dd6daf94aa72a2a4e4456cc3ca604fc5be258bf2b7d3:action", "state_id": "5b263edb703475bce54da7a3c8c41dcb7ad89ef22276c4ebb2f25352090640b5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.265625, -1.11328125, -0.8671875, 0.888671875], "student_probs": [0.08146055042743683, 0.09486573189496994, 0.12133511900901794, 0.7023386359214783], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a610179d26450b1e387e777d6a19868aea582fdfad164b886c8ac3856058073e:action", "state_id": "c36177c525acd109268ddb6cca53182822910f14dd3d1fadd4ce8a3b6eae691e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.19921875, -1.140625, -0.7998046875, 1.150390625], "student_probs": [0.07126177102327347, 0.0755620151758194, 0.10624779015779495, 0.7469283938407898], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a9a6e625499384d18fed3d92ee59405ca56facce7c403429bcdfa5a25beee054:action", "state_id": "2949ba6b8a441ef50a8320a5b53785c1ab18aa2653f5f9ea3c0484383e19a4bf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.45703125, -1.48828125, -1.140625, 0.8671875], "student_probs": [0.07374568283557892, 0.0714767724275589, 0.1011929139494896, 0.7535845637321472], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "29c65d3c373b1989c4f6362e1150318674d020c87e1d57b311416d4101e13187:action", "state_id": "d32f21bfbb9fb9c8f3c259cba334cc62d03325cf6807f16cfbc0d000d6d87e60", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3984375, -1.34765625, -1.140625, 0.689453125], "student_probs": [0.08761118352413177, 0.09217508882284164, 0.11337729543447495, 0.706836462020874], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a1fe861a466f06848959b83e0a22afc9626409111fdf67022556ddb031d87268:action", "state_id": "0f90db3f6b6a46d2885d2eb4334bffef172d0202cfe1fa5fcf9a2d096ccb20fc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.41015625, -1.3515625, -1.11328125, 0.60546875], "student_probs": [0.0916472002863884, 0.09717759490013123, 0.12332478165626526, 0.6878504157066345], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "37693aa4f98ba85f2e03e182d83ffa2d17692c1d35f805ce46df9214b46651b1:action", "state_id": "6532b7508002551d73e9b7ca875dc3aba828b44a6dde93b4de8c9d53e065a995", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.40234375, -1.35546875, -1.09765625, 0.58984375], "student_probs": [0.09314676374197006, 0.09761697053909302, 0.12632574141025543, 0.6829105019569397], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "682d35d27356596bed2f4eb2d216345faeae025f85afb33813e55161d8ef48ac:action", "state_id": "103333bf53c6555651424c2d1e89c148075793f24cf3e21e517e2cc7623ccd73", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.578125, -1.50390625, -1.21875, 0.63671875], "student_probs": [0.07893000543117523, 0.08501096069812775, 0.11306200921535492, 0.7229970693588257], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "479bd00eb8ec72f329fb08c0d7ca5fe1fc8be58b20671a107541db4f262834fc:action", "state_id": "63684295c75ec385468bd82da0b73ec8644b7605b2d7033528f0867817de9b9d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.52734375, -1.453125, -1.171875, 0.6875], "student_probs": [0.07896479964256287, 0.08504843711853027, 0.11267087608575821, 0.7233158349990845], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "19c18ae2cf42c0ee0b0174e5d5e203c29a92be68458bc154ccf95247534d0b5a:action", "state_id": "e923c31fac6f242fd24757de5bed1a1fd0f02a59f0a1337493f686c6b40f2a40", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.390625, -1.296875, -1.044921875, 0.833984375], "student_probs": [0.07836291939020157, 0.08606483042240143, 0.11072548478841782, 0.7248467803001404], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2904097b032385d2e2d7558fbbf49d88b3793968147975df1d3c3dc8bb513554:action", "state_id": "a7f819b187585f373291c186abc752d5eaf731c8b2554f69fb76745a24a3c811", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.27734375, -1.1640625, -0.96875, 0.978515625], "student_probs": [0.07677574455738068, 0.08598475158214569, 0.10453087091445923, 0.7327086329460144], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e9239cbf7acf2c301ef345fd9c1b7fc3d8105d587286cee04c7067a1ded2cbe1:action", "state_id": "84f471f4cc1d7f93fc45e660d5c7c078f961b81ade7e08a53e35d7220349a040", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.29296875, -1.14453125, -0.98828125, 0.98828125], "student_probs": [0.0751589760184288, 0.08718594163656235, 0.10193068534135818, 0.7357243895530701], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "807bb10e25e3eda91276738fde8cc78e3daa080bb8b19b8e8c09d81f34d3b54e:action", "state_id": "f06a85a07c25835142bed514a76a86f1fe95823d3c4451da4ec1ebff50f66590", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.421875, -1.3515625, -1.11328125, 0.720703125], "student_probs": [0.0836419016122818, 0.08973465859889984, 0.11387921124696732, 0.7127442955970764], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0cc63b0e44112b5b91b778d50834c1d618124b63148754a3238c6e4133081706:action", "state_id": "aac9cdbc835c1d86acae8c5a20ec530a8f1dd28889f274fd2550c7932c8fb507", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.421875, -1.33203125, -1.11328125, 0.669921875], "student_probs": [0.08654285967350006, 0.09467817842960358, 0.11782889068126678, 0.7009500861167908], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9297fb0794a1faef3cad57ba72253ecf215ab6e901dfa70289ef3f70f7ce9ef4:action", "state_id": "5471c782d4e75c82d20902f8ac6fa07e5cf617a0892d6d71eb5b86608114f254", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.38671875, -1.359375, -1.01953125, 0.607421875], "student_probs": [0.09244470298290253, 0.09500737488269806, 0.13345950841903687, 0.6790884137153625], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "49502faad372b3c071432a1a154cacad01c0c8a9f89952d125746459203ea692:action", "state_id": "a472131837096204d003e9c74478a1fa9dcbc36093b6bcd7763ef7bff25240f0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.53125, -1.4609375, -1.1015625, 0.6328125], "student_probs": [0.08119481801986694, 0.08710931986570358, 0.12477834522724152, 0.7069174647331238], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b48ad523fb639fe6e586f0b05ff339fce6b8b6e3a12253004dd9965ee3beb4ef:action", "state_id": "78e854919c936ecc9448aebe08b2588834a29f4bfb5b4573ad39cdf1ff5a62b4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.66796875, -1.50390625, -1.32421875, 0.521484375], "student_probs": [0.07987958937883377, 0.09412115812301636, 0.11264827847480774, 0.7133509516716003], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3fecc190549b999e51137166e385df921f884202bf15fa9a6d0e6048972ad613:action", "state_id": "c086c9d3bd8d039ee9fae8742f90c5b76ec7e20fb5a7ad8e0b48f693efe04a4e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6015625, -1.51171875, -1.2421875, 0.5], "student_probs": [0.08543083071708679, 0.09346160292625427, 0.12237400561571121, 0.6987335681915283], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "137f9bcfaa36f942e700ad0e94cfbd79c99c27eefd126bc8a7ec71e0fb514d29:action", "state_id": "d528c9762aea71110ddfd7a0cf0549045d057b8bf61e90bda431c50a1939f9c7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4140625, -1.34375, -0.943359375, 0.556640625], "student_probs": [0.09216859936714172, 0.09888246655464172, 0.14757294952869415, 0.6613760590553284], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "146ba54ae663bc96a77a7681c8d082c37e48627f1d238b44062e8cde3061f279:action", "state_id": "f6937e3d499b22c2cd43a260253d87b10af15c1364dc3db10e3a7919d2f91fbb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6796875, -1.76171875, -1.5625, 0.458984375], "student_probs": [0.08670221269130707, 0.07987382262945175, 0.09748192131519318, 0.735942006111145], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a0a9dd4bad5429433597a7319c20b9c3043e6c8a823458f4e718a14e30cef1f7:action", "state_id": "6b22f0ceb1389cab3a401099368a0355a3592c3318e0da0d2c7f9750146f6ef9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5, -1.515625, -1.19140625, 0.55859375], "student_probs": [0.08943784981966019, 0.08805124461650848, 0.12177044153213501, 0.7007405161857605], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "32f0d0145d3ab060e7d99b4c50526f01b0aa88a61c09ae538a288ff72a0a2b6d:action", "state_id": "c4c8d1f25f25c9ee07663bbd29a3df3ee6d3cdaad184b63ec1c58a59e72d5b64", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6328125, -1.6484375, -1.37890625, 0.419921875], "student_probs": [0.09039369970560074, 0.08899227529764175, 0.11652208864688873, 0.7040919661521912], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7e0813829d563b9aef0f0209e46c07edb2f1b79bf11150cea5b3fe7b7509f0e1:action", "state_id": "87d49ccd8cdfe99ef62322345da25b0bd6857c934241b0c5f8360c91284361ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.50390625, -1.36328125, -1.21484375, 0.50390625], "student_probs": [0.09146471321582794, 0.1052752435207367, 0.1221214160323143, 0.6811385750770569], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "109a841965b3a1ddf24945999b4e9fb8e4692b31dbb06c64c21918641ac8ab7c:action", "state_id": "7eeadfcfd98ec3754f95dd5eae824b0300583523fe734529e3a465ff929f7359", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.11328125, -1.3984375, -0.255859375, 0.537109375], "student_probs": [0.10731931775808334, 0.08069305121898651, 0.2529597878456116, 0.5590278506278992], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fe0b390c0d4dfc4e1846b010bad547d4cc91cb9aa469b651a07c95f66a5b1442:action", "state_id": "c4f3710a7fdd93fb25369abe0a985a114e8b24089947ca329d2a71b287a8fb37", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.98046875, -1.03515625, -0.08984375, 0.8203125], "student_probs": [0.09580555558204651, 0.09070686995983124, 0.23344479501247406, 0.5800427794456482], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "97907ea33fc39a2173c1411ac0a534bccba641da2fadbd1c37062b06c6c1ce18:action", "state_id": "f019307d4c79041ce5cf7e36d32984aa8837775ccdda6bddc1b8c9f240f4cfeb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -1.037109375, -0.0546875, 0.74609375], "student_probs": [0.0939972847700119, 0.09418104588985443, 0.25154978036880493, 0.5602718591690063], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "84085c2631345298ec3cf9123da245c816975ec4de1d6b70dddb59eca95596b7:action", "state_id": "7ed250bbd7fd5fb89d9622fc5b2b6929e1d737c3e08baaedeede2f5bb673c56f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.021484375, -0.9453125, -0.009765625, 0.7890625], "student_probs": [0.09138043224811554, 0.09861301630735397, 0.2513258159160614, 0.5586807727813721], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fe4bba68b1f78b1d9edd379d04f73c3a53dd9d033bc62f06c24d1b2dcfcb1445:action", "state_id": "604d662f107af8f5329d0e7e8f1fddf75dbef2b300551c91f2db541d4e5085b2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.005859375, -0.9296875, 0.025390625, 0.798828125], "student_probs": [0.0912259891629219, 0.09844633936882019, 0.2558496296405792, 0.554478108882904], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ebad2f34e59a2df67ea7837ff488724ee22ff362f2dbbbfa141e809ef7f645c9:action", "state_id": "13d7c9764bbf7796c27e10a4db0ae017ca9f6177a48f936153e724ead99337e0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -0.91796875, 0.013671875, 0.814453125], "student_probs": [0.08868718892335892, 0.09913112968206406, 0.2516613304615021, 0.5605202913284302], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9c6972026c0246ee93625f5644006b64e1839c12d8eb100f7227fbb1ef9f46a4:action", "state_id": "c04fff4158654ec691f07f187b05ec62affc51e70f841406120e03d69d1c2f2c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9921875, -0.908203125, 0.029296875, 0.83203125], "student_probs": [0.09039240330457687, 0.09831184893846512, 0.25104808807373047, 0.5602476596832275], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1099b9022ada7825da18fe80958826c9198d0df80605fc7d419b6a6d69de89af:action", "state_id": "fd1a535252c8dbc4c5af700b30e51d68ec3d6a636e01c8c7c759fd1dd8bb4942", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9921875, -0.900390625, 0.029296875, 0.849609375], "student_probs": [0.08943489193916321, 0.0980333536863327, 0.24838879704475403, 0.5641430020332336], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fe0624af38f4cf00a0a6c2745c1e058883ba7bab84053656b1969cc5f5c2bb14:action", "state_id": "9a0c7f49ec4957215c6bfb1d7b984f910011763b6a4303c3296020585d893956", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.998046875, -0.912109375, 0.033203125, 0.857421875], "student_probs": [0.08858178555965424, 0.0965309590101242, 0.24843376874923706, 0.5664535164833069], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cab6031b733dca9e41060e99e49c1a753ad434c91389c6741c95d9946cdf7c44:action", "state_id": "0401f20998d36e063255a263c77916964aabf82e13c97354c6d50eb8209f2be6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01171875, -0.919921875, 0.025390625, 0.83984375], "student_probs": [0.08859783411026001, 0.09711582213640213, 0.24993896484375, 0.5643473267555237], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "760e4fccfb9d46bf13768efe01d0831629f9897dcd66daa4af3c09849aa14f70:action", "state_id": "0b0f4e9eb815fbacb71b61459dbad5d315be4b1e441bf22a205fee25b6d383ec", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -0.9296875, 0.01171875, 0.830078125], "student_probs": [0.08915895223617554, 0.09715991467237473, 0.24907758831977844, 0.5646035671234131], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "032a0421c184ddf20d01832b43c3c919bc3fd4508a5bd47956c940f36a3a46cc:action", "state_id": "159bbf168d784104d9704753a4ac0fb9dcecd161663962ab7c6b20ac7a8fac59", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.02734375, -0.951171875, 0.021484375, 0.7890625], "student_probs": [0.09022688865661621, 0.09736817330121994, 0.25753501057624817, 0.5548699498176575], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b449ba3b82e5469c6bce103c75e0db164edd621f7bc1386ca61ce9bee3fb4a9f:action", "state_id": "5e812987fa395b00251e7f451040d249056b8fdf9d014ffea7719ab324e3fbf7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.037109375, -0.953125, 0.005859375, 0.7890625], "student_probs": [0.08980442583560944, 0.09767235070466995, 0.25483161211013794, 0.5576915740966797], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a48a49cf0677e5ed37b1a5c55213ccaaa3dfdec819ff1af5c7e830558d300d12:action", "state_id": "b04fbf4382bb2e2f591b171b5abbc44d4ccf386cc304ed50d90e0d264e7fda75", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.951171875, 0.013671875, 0.7890625], "student_probs": [0.08992738276720047, 0.09761524945497513, 0.2561792731285095, 0.5562779903411865], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6998d0afb8e4448d402c9969b5c84dcd6cfb83a30aacc7fc6da7c2340be59a25:action", "state_id": "4bb25b849ebbc9966ce38878ad4d92b6477a25383f1d19e474b5b5ecdb8d0b4f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.94921875, 0.033203125, 0.798828125], "student_probs": [0.08961044996976852, 0.0967029333114624, 0.2582855224609375, 0.5554011464118958], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5e2ffbcf5a870321c44f534804ceb2781d6ad8c3bffa55a7d6b45d283e524c71:action", "state_id": "17d354e7097bd08e87d76af757da6c3be393dd57c26af7862b903ec785c340d5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.037109375, -0.96875, 0.03515625, 0.7890625], "student_probs": [0.08926331996917725, 0.09557870030403137, 0.2608267068862915, 0.5543313026428223], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d6723eb7a23bd8c430f54a379a54a229bfbd7abf5ea06e08938e15a7f8cfea9f:action", "state_id": "5a7c5cad7fd30a27cd6d8af58348ddc26f826e03cd9b14e279b177242001ab86", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.037109375, -0.9609375, 0.04296875, 0.796875], "student_probs": [0.08863023668527603, 0.09564514458179474, 0.26100802421569824, 0.5547166466712952], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "740d71f4dcf43166d0e85d7c22775ca08028a0606d5f49d9292169f41ceace73:action", "state_id": "cbf1ad359a35b66a427f2225dc3814b95e9786a4bb6a565434f448d803dd3923", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.951171875, 0.021484375, 0.7890625], "student_probs": [0.08974706381559372, 0.09741951525211334, 0.2576708197593689, 0.5551625490188599], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "91ccfd80e266b31021de0ac1614f856dc2ff8517c6f351ecb70dd9e0e54cff56:action", "state_id": "6f6d56f847ea42e9bb9ba097504e53ed62fab27e44ab47dc989fb759b52af203", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.046875, -0.958984375, 0.025390625, 0.796875], "student_probs": [0.08822967112064362, 0.09633521735668182, 0.2578063905239105, 0.5576286911964417], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ee12cfff8ec52d9d2b519fbb8e990988c1744d7f1e9975cd9702f161bd98e24f:action", "state_id": "df42764658e2e61219d5ab8bc3249acd469ded553e67eac4a31c66556d336339", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.05859375, -0.96875, 0.017578125, 0.8046875], "student_probs": [0.08716662228107452, 0.09536056965589523, 0.2556970417499542, 0.5617757439613342], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a56bfed69d2fa4cf0225bdc40dc1848c1eec32aebd674e023b9a54a4ea134cf1:action", "state_id": "211d676ffafbe5703cb2087b79a881ccb47b08bbaf22d275436ad9e859e26773", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0625, -0.974609375, 0.021484375, 0.8125], "student_probs": [0.08643697947263718, 0.09437782317399979, 0.25554534792900085, 0.5636398792266846], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b404370ff60036b078dcd2dda4882099df763db7499fc428971705c5e0f1fc16:action", "state_id": "0175b3db65ac4467f396363dc476bd88ee39fc546c33bfd44e5b26e7921b023c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.08203125, -0.9453125, 0.009765625, 0.794921875], "student_probs": [0.0857655256986618, 0.09833065420389175, 0.25554895401000977, 0.5603548288345337], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0050f61c63a53fbb3e101e6dfb8daa8d43ac925a91f0000f84dd96d4a882e0ba:action", "state_id": "b94f701936e13cd3dddcf79f4f7c0b5e37194014d8927366d9f564f8d7c509b3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.08203125, -0.98046875, 0.0, 0.810546875], "student_probs": [0.08551377803087234, 0.0946551188826561, 0.2523226737976074, 0.5675083994865417], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e0d8edfa726559279c4843ab1ac3f7b7a2a7b246323015563cb277a8a364ce98:action", "state_id": "e7661e029c6d3053ef5d3556dc0b53bb24a34f126f12fb6252485e3e32866014", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.052734375, -0.970703125, 0.021484375, 0.828125], "student_probs": [0.08641299605369568, 0.09380041807889938, 0.2529917359352112, 0.5667948722839355], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "07f90ce5c664fe0641fb4d1cbcf83b54428ce87fb97960cd0cacc9d44946e607:action", "state_id": "8b54a9e04e02bc8128d3cd9f54236a9b73d115be84fc9f51de7a9e099310e567", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.072265625, -0.99609375, 0.03125, 0.818359375], "student_probs": [0.08534315973520279, 0.09209790080785751, 0.2572879493236542, 0.5652710199356079], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "919e66a1a51648a70312b2f82eae33cc855b0542340c7b4d08dfdf6726393c01:action", "state_id": "09043dae7bc68ff38f396f003f356e9f0998c3d781b8c6b042f6715d971cb2bc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0546875, -0.982421875, 0.0390625, 0.828125], "student_probs": [0.08596713840961456, 0.09240958839654922, 0.25665047764778137, 0.5649728178977966], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a9302f8f295ea67c2d436c1f1866df426f1cf30888739a2d35fd3eaf01948122:action", "state_id": "bccc8a0c49ce7105227278501b830eb762b872d1edcaffe53235029904922bd2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.076171875, -1.009765625, 0.021484375, 0.80078125], "student_probs": [0.08621163666248322, 0.09213099628686905, 0.2583877742290497, 0.5632695555686951], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "280d921c14675bee813fb8d4cdaa326096755f2f76776c44b4f9afc06231d480:action", "state_id": "349661c313688c5ad4045af48c43f32820eb6b32092c9e9494123742360725bd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.037109375, -0.97265625, 0.009765625, 0.8203125], "student_probs": [0.08831965923309326, 0.09419959038496017, 0.2515992820262909, 0.5658814311027527], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "40df8f6db2433781fab57817442c9a8c7f6500698f940bb58b31d8c47e644a75:action", "state_id": "58342a407c10ae6a4db2c8f9b1af26320e74b40f65cf60ef8adda38fd2ddcc2b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -0.98046875, 0.03125, 0.8203125], "student_probs": [0.08759111911058426, 0.09305832535028458, 0.2559405565261841, 0.5634100437164307], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e3e1698b6d944371a501676cad1a46c01ff8e66d23c6c82f98ce2917e0b8814b:action", "state_id": "0ae85f722c04bde5e8ab6e51cd0bea678d4a34213a778dd201970ed02c5a114d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03125, -0.99609375, 0.0390625, 0.8203125], "student_probs": [0.0883248969912529, 0.09148529171943665, 0.25758105516433716, 0.5626087784767151], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "94aa0696fdeed36ca4a3c3be97d9d058066105da017ddf2fa1e0a06fe9853ebe:action", "state_id": "289b3bbaffd490785d7ee76f58f2344f66c04a267460bfb3e679a43738e5719d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.046875, -1.0078125, 0.0, 0.810546875], "student_probs": [0.08852873742580414, 0.0920553207397461, 0.25219491124153137, 0.5672210454940796], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0bcf954973d3fbecdf02a5484cc45f24382d2e6d244854c9eb3d05f26c07108f:action", "state_id": "195ff515a711c9688f7671613ddbd8696d4968f2030c5b9d42f1ed1873e89996", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0546875, -1.029296875, 0.0, 0.78515625], "student_probs": [0.0893467515707016, 0.09164436906576157, 0.2565214931964874, 0.5624873638153076], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fcf728615fdf0fc4c9e9cbed359173672d2eb1cc2fbaecc4262d1221ac99d65f:action", "state_id": "8dd33d57e9c1ca3d909592e91252c3420099364d9a1ebe77cc7aa54a46c1704b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -0.974609375, 0.021484375, 0.8359375], "student_probs": [0.08792334794998169, 0.09286556392908096, 0.25145062804222107, 0.5677605271339417], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "31a9860b2ea822c2083a9a0a9a48090cbef0dd5e19453fcc608b9c54ec4e2d27:action", "state_id": "984f73a808ef018e7231e868e3a326e50a8ecb84e87cf2d02266473df0f5a23b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.95703125, 0.025390625, 0.8359375], "student_probs": [0.08800563216209412, 0.09423203021287918, 0.2516859471797943, 0.5660763382911682], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3dc3656f6b0701d3efc5727abc3d9f867fd0e55173ea4ce0410cac42839e2794:action", "state_id": "9131b5e3b6b536e85be41b5dd521463cda0fb663e28630a375dff85919a8082d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0546875, -0.994140625, 0.0, 0.787109375], "student_probs": [0.08895722776651382, 0.09450971335172653, 0.2554031312465668, 0.5611299872398376], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c080358338b49bdabd72963f4f75ca0caedcca6dccbfc9d3a6f0c0408a85bcaf:action", "state_id": "c44b867610f2e3c20e2ec3a1ee4593ed6c13ded63803662934d0326d68a6311f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.048828125, -0.98046875, -0.001953125, 0.802734375], "student_probs": [0.08857984840869904, 0.09484687447547913, 0.25234049558639526, 0.5642327666282654], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "70b22d137969c00aa5cce9e164e660aab744d4af19e92d2be54d930a3198d0b6:action", "state_id": "a8d2cdce5ce34f0c13d04f4fce548a91eff887c7b50f1743fb0f70428513bad3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.02734375, -0.9453125, 0.00390625, 0.814453125], "student_probs": [0.08930228650569916, 0.09693672508001328, 0.25045445561408997, 0.5633065700531006], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7dfdfa90506da67a3c0983959e6918ea9c110b014ecd34056864bd71882ffb29:action", "state_id": "43c14e7c859eb58323ec114d8018cda9c36d7260e760ecb9ee040811b7c2f2a7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.02734375, -0.95703125, 0.021484375, 0.830078125], "student_probs": [0.08822742104530334, 0.09465420246124268, 0.2518278956413269, 0.5652904510498047], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9b3b4a4d4ca3186f90a1b9e4b29291fb66c1bf7f4592a47d4019406d3d5ace4f:action", "state_id": "655f488a1bba98acc3b199071ba737c48c82f76fe0ad298a5396950dfe7e18c8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.93359375, 0.021484375, 0.837890625], "student_probs": [0.08717472851276398, 0.09630534797906876, 0.25028544664382935, 0.5662344694137573], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bd91b9a3eb4c1debbeb02ed6295a16d1ebb1ad4676ffc00618d2c985d8d0c2bb:action", "state_id": "d1d55be54e04f940f7245a5d553c8b77f47ebab168f28a283e11b22b40041dab", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.044921875, -0.9453125, 0.005859375, 0.830078125], "student_probs": [0.08706673979759216, 0.09618604928255081, 0.24900081753730774, 0.5677464604377747], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a6694e08f049a538e4bacaad60401c03dfd69fda7f58cde03836e68c2802cdce:action", "state_id": "8af3c44c8be3f87b515d31b0657ed445eb1d6d5e386032f9c45cb2164fc8bed7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.05078125, -0.95703125, 0.013671875, 0.828125], "student_probs": [0.08662599325180054, 0.09514004737138748, 0.2511506676673889, 0.5670832991600037], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "faaf9fcfb387a4e174a3a5c59a372772f611a21aba181a6ec8c43c249570f940:action", "state_id": "a949daa26b6e614583794ce263a8b31130c6d9903f1b7b0c922a0d62a50baf25", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -0.958984375, 0.005859375, 0.8046875], "student_probs": [0.08875653892755508, 0.09634431451559067, 0.25284385681152344, 0.5620552897453308], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e3c601fd19336182bf04d676ab4f3b73f817a40fc706a358bd3e0f01df0803cb:action", "state_id": "2548b57be507b808172d92a8baa82f616d161b8634ff08491dbd9aa63acab8a5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.052734375, -0.951171875, 0.005859375, 0.8125], "student_probs": [0.08736170828342438, 0.09670059382915497, 0.2518039345741272, 0.5641337633132935], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "90f1eb08ea6316f877e06b698db98fe459b04e21001db658a8586abf34fea8a1:action", "state_id": "6640d2ccdb0e74b4bbcb9b776bd44d859a87c079e5dd12922d5a0cfaef69e82d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.05859375, -0.98046875, 0.01171875, 0.802734375], "student_probs": [0.08749042451381683, 0.09459970146417618, 0.2551475167274475, 0.5627623796463013], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cbde3ac7890421445d25c05c9da9c2ace64654e5ad93395d7aa8e751a748a3df:action", "state_id": "bfbe034dd4d5dd59c2561b7e5197e5e2a26fc5645c95ed0cafcc7a94ed19b3be", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.056640625, -1.01171875, -0.029296875, 0.748046875], "student_probs": [0.09159478545188904, 0.09580320864915848, 0.2558824121952057, 0.5567196607589722], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d18299d69e9592ae3abeeca51dcf233e7d7fa5feef05b24e4751f6342e019e9c:action", "state_id": "473df85577079655547cb6f67321db2e811f94ba30506b6e295e02c6e22db220", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.048828125, -1.015625, -0.037109375, 0.763671875], "student_probs": [0.09166047722101212, 0.09475498646497726, 0.252096027135849, 0.5614885091781616], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d7d8ae5e123d210b650fec9f5e218274c47a2ef97e29631e732630a0ec9112dc:action", "state_id": "2e6a5f0331fb19f32a57639f3e75af7683eae3e06a899f5c54e56ee2c69494ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.05078125, -1.119140625, -0.072265625, 0.712890625], "student_probs": [0.09589338302612305, 0.0895572081208229, 0.2551247179508209, 0.559424638748169], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0d5fd3c581bce525779dabe8d7f57ca95b4e77f816688677d649fc7d69063c5:action", "state_id": "d02f273f58cefefd1ed421af338718bdbef733489b0f00731a0b1ad082e8faab", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.048828125, -1.099609375, -0.02734375, 0.73046875], "student_probs": [0.09386596083641052, 0.08921832591295242, 0.26069527864456177, 0.5562204122543335], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5dacc8315cfdca5a1d369fccbb95f60c45b895e7aa1111f0ecd39dc57314f823:action", "state_id": "0bb33e4eee4e9ed102b251e97efa6ce59ca1bf7f5a9e442c8c71f3827522a56c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.125, -1.25, -0.30859375, 1.267578125], "student_probs": [0.06628435105085373, 0.05849573761224747, 0.14995872974395752, 0.7252612709999084], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d90c437a27566de7317f6792abef8f975405a15be50d682b65f94bb7e408d569:action", "state_id": "2c5469fb36a24249d597e636e3ca431b22f6b067e393a2b7475880e0d77a1479", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.060546875, -1.19140625, -0.259765625, 1.123046875], "student_probs": [0.07702512294054031, 0.0675773099064827, 0.17155654728412628, 0.6838409900665283], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0fe0e64ff9394d073950bd104d811d98017877fc499b2211243f237df84db2a5:action", "state_id": "7c611f9950c516eb7fdbf9f06ce5924aab295a18b1415dd297ba40495bce1180", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.19140625, -1.2421875, -0.330078125, 0.603515625], "student_probs": [0.09675117582082748, 0.09196069091558456, 0.22894242405891418, 0.582345724105835], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "52a8a26a2ddefd2316a9665603dd8275a8da15d80c368e40b15deaefe4f04382:action", "state_id": "350d1433c5f38520752682e686b0e43dd310947d43a2e604c42349eb466727f6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.142578125, -0.9375, -0.23828125, 0.751953125], "student_probs": [0.08812710642814636, 0.10818669945001602, 0.217691108584404, 0.5859951376914978], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c83bae76b26d19888a55fda92d34433d8e6b2cde9158b3ab38ffd389b373006a:action", "state_id": "dd17d61fe23cd47fc0576db359e69c29d70889c14f068c56bec79f9c0e218484", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.990234375, -0.919921875, -0.103515625, 0.77734375], "student_probs": [0.09655633568763733, 0.10358982533216476, 0.23435695469379425, 0.5654968619346619], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "096d9e85cf1884dd5d7f7c7668ea802a9d950421cc0236d81c3a965958389a4c:action", "state_id": "f4869ad46e77a950bd12b314aba9421b22403adb1d82212c62f293a3cc9009ca", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.12890625, -1.42578125, -0.2734375, 0.513671875], "student_probs": [0.10794366896152496, 0.0802169218659401, 0.25393497943878174, 0.5579044222831726], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "535db078b2f6b7b47b4a850fdcbeb0cce0df8bb38890be9b23a73becb35c750a:action", "state_id": "6cb9fbb7fd49d73c39df2ad9eabed9ec5d78c03f9855b053e24279e0177ed9cb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.986328125, -1.05859375, -0.06640625, 0.8046875], "student_probs": [0.09583210200071335, 0.0891510397195816, 0.24045178294181824, 0.574565052986145], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ad3caee9c76de619b98dc1ec2f159369d42d945fc957dd04774c5ecf8fd42e0c:action", "state_id": "730d381a6935764d2f949f503a6195676ad651d3187bb11a7f32cd47b140573a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.046875, -1.052734375, -0.048828125, 0.751953125], "student_probs": [0.09302587062120438, 0.09248238801956177, 0.25237712264060974, 0.5621145963668823], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "215de9de06178f3ca2dd795cd251496cb4b3f4d9a94e64542d7c57a97799345d:action", "state_id": "fcc07892241a9cde2ba462cb9f88c289f596eee1e4d977ce48b08de7ffb74f2e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.955078125, 0.009765625, 0.798828125], "student_probs": [0.09020109474658966, 0.09677164256572723, 0.2539653480052948, 0.5590619444847107], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e512e7dae82e37a150043d9d087d8591c0c71bcdb90f026bb1ca5f54026a237e:action", "state_id": "d5a5ea0c1edf6c56d704ff164310ed81ef5b535c80fb8a2694edb9a8fc54a23f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.013671875, -0.919921875, 0.0390625, 0.83203125], "student_probs": [0.0885244756937027, 0.0972251147031784, 0.2536647319793701, 0.5605857372283936], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "eeadc1e735b27d9b566f9a0cffed010dea0d0ec28757dc697c424e3868d460ea:action", "state_id": "c84c35b0075ac207a7dc4e98fd92a1beeb76e4705def77d0712ea7e51a1b1bf5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.00390625, -0.9140625, 0.048828125, 0.857421875], "student_probs": [0.08778280764818192, 0.09603467583656311, 0.2515394985675812, 0.5646430253982544], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2ca9bcadbd5d3accc64d5557d386c97c12c386539c4eaba1109f0ab76318ef23:action", "state_id": "859cc0b5796b0232ff86070a20dadfcf3f68a436414f6d0d5b66854176fa56c0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -0.88671875, 0.056640625, 0.849609375], "student_probs": [0.08682797849178314, 0.09877406805753708, 0.25371065735816956, 0.560687243938446], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "84004405b8f9ddd4ba8d00baa3c5f9e8360c7a0a4a89f39042e6747537c3a698:action", "state_id": "659895da3b40f166c8309d32ec985a280f6e01c5ef392b64ff5d867a59646535", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -0.8173828125, 0.076171875, 0.841796875], "student_probs": [0.08509132266044617, 0.1051764264702797, 0.25703027844429016, 0.5527019500732422], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "957092ffca7acab56ca13dc7d2e77ae16e88b4560a7ad47571153fb122208adc:action", "state_id": "2175d9ae9e81ff14656cea1919de0fb318e7ecb2d50ddc706585a7addb9ce956", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.11328125, -1.38671875, -0.224609375, 0.6015625], "student_probs": [0.10258025676012039, 0.07803893834352493, 0.24946467578411102, 0.5699161291122437], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "970de74fcf1435f3c82a69936880190d76e5a53d8b011c9231cebb0ab03e530b:action", "state_id": "bd7e6b931a4a5b9e567dabc72e331fe6eb2e6256467525dfa161e1205e9a04f1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.970703125, -1.033203125, -0.033203125, 0.853515625], "student_probs": [0.09353702515363693, 0.08786990493535995, 0.23885515332221985, 0.5797379016876221], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "94c325d8e6a1f73b1bda4098ff3271964998d30e9efb09c28e0842a4857f4895:action", "state_id": "5e84a187cb087076f6e9240ee4dd9713ce1446bdbf65d6d07c1d73049f957dbd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.052734375, -1.037109375, -0.03515625, 0.802734375], "student_probs": [0.08947000652551651, 0.09087894856929779, 0.24751758575439453, 0.5721334218978882], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3ee8fb1687e5fccf52655f26623cd412bcba142395a1ff678f17ae76b0501f16:action", "state_id": "954ad9be07fa5286709842bc7331c1a7dbb6fc8239a025a917d90522599c2677", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -0.958984375, 0.017578125, 0.837890625], "student_probs": [0.08684969693422318, 0.09427445381879807, 0.2503281533718109, 0.5685477256774902], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "216d28c00644ca127af97599e0533654b749aed73e2bdc1aa0ccc29fd938fca1:action", "state_id": "ea256c6b1257e687423951d8f51ade0bcca967b4efce2c80ef5c9c9facc1676a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.9296875, 0.025390625, 0.83984375], "student_probs": [0.08758281171321869, 0.09637895226478577, 0.2504767179489136, 0.5655615329742432], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4210f6d06210e565f7d355ea700a87d3ade0ee8b39a41a303c7e3bb594f2a955:action", "state_id": "0ceb7d8b32784ed881c858af4601a8aa23ac2c548656c39b8da0d40a445ac35b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.931640625, 0.033203125, 0.837890625], "student_probs": [0.08690197765827179, 0.09619172662496567, 0.2524434030056, 0.5644628405570984], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ffa7ae8dd71dcffab6378c167e47751e3003d2d52a03850e3e2e3495d4ea2780:action", "state_id": "d104af853b77581ccbb7d72aefc56ade1dba7e2a3c7e52f72d228139d317e25b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.99609375, -0.916015625, 0.05078125, 0.857421875], "student_probs": [0.08838354051113129, 0.09575222432613373, 0.25178125500679016, 0.5640829801559448], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a5675db2ab37c25b71694e58a9ab1a35e14515d5b9ccfca95d1cb052c08692fa:action", "state_id": "f887e4275047c78d1cb45f96e454dbbb22d027fb9e74bb0b5ae3602d68449848", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.005859375, -0.91796875, 0.033203125, 0.83984375], "student_probs": [0.08888110518455505, 0.09704648703336716, 0.2512282729148865, 0.5628440976142883], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b197f3f529d0dab9a4c594bf681a546c02b9d401c5c705303b1217ee2cb32f72:action", "state_id": "57878ae584f6f0779c2a1449dea510659e98c2fb16f2e71ce78e0fc35ec0f75a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.00390625, -0.9296875, 0.033203125, 0.82421875], "student_probs": [0.089925616979599, 0.09685369580984116, 0.25368472933769226, 0.5595359802246094], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4acaa6c953b01565e4144448d8b7fea212ac13f527ae8300be7a9225ef6b54c2:action", "state_id": "6fe3a0414f58a1d53470127dd161a5256cc710e5440d533f19ed44e7550a9e76", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.017578125, -0.94140625, 0.015625, 0.814453125], "student_probs": [0.08980076760053635, 0.0969083160161972, 0.25234484672546387, 0.5609460473060608], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "10e0c29b79aae39d49ac7160ff7a409b56e01d66689c706db110464feed7b3e8:action", "state_id": "8bc9362fffbf52abf53e54e048b33a22c0e3c109d3c0df2661b7e2675acd5747", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.951171875, -0.009765625, 0.796875], "student_probs": [0.09006892889738083, 0.09776890277862549, 0.25063878297805786, 0.5615233778953552], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c5c986e2bb3f7f6f37dacc6d0de9468ca173f7d870446c9a413928fca50144ca:action", "state_id": "b6f2bf0243420ae1bb1cbe2e36774a52a8b2aaa37c56d3c1aec662a51bca6150", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.97265625, -0.01953125, 0.75], "student_probs": [0.09353669732809067, 0.09860166907310486, 0.2557532787322998, 0.5521084070205688], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7224b5dd0dc2b704d13d9f424ef728cc43994000071512a1031c44aced28bf6c:action", "state_id": "feb5ee9386cc4dd2b2fb01aa5d4c46441f74aae584dfbfa7276d14b171399e76", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -0.92578125, -0.154296875, 0.83203125], "student_probs": [0.09092044830322266, 0.10142908245325089, 0.21938851475715637, 0.5882619619369507], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "09566025dc4d672c16f35d458283eda9b1b7833c37a05751794efe4599a068d8:action", "state_id": "cdf4af3e7c92bd303962a4685dfa3f4618fac5ca20a37892ec48e5f20a02f120", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01171875, -0.912109375, -0.23046875, 0.841796875], "student_probs": [0.09371054172515869, 0.10352571308612823, 0.20468264818191528, 0.5980810523033142], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5b1061b1a9ec26e2a67f4a0cab84e686ffe81123867e158e5577b2e641bbd8e0:action", "state_id": "8ad4b0cdc289903da819da5365e1a0dede20bab620d3a8223dcaef5c9db51d2c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.00390625, -0.9140625, -0.263671875, 0.833984375], "student_probs": [0.09547723829746246, 0.10445241630077362, 0.20016103982925415, 0.5999093055725098], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7d66e3684ddeed95285b34f3584e0eb4a268b5716edba4dc95ee1052c922d8dd:action", "state_id": "4a52cfa4f6571a545586ed38a4ce16ea00ba6caa36c7dd6c606766f810d83099", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03125, -0.9609375, -0.3525390625, 0.794921875], "student_probs": [0.09752056747674942, 0.10462429374456406, 0.1922457069158554, 0.6056093573570251], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ec2be3f28577d09fe0d88a58c3a75b3eecdf39bb6fa5433955ecc19fb3fb02dd:action", "state_id": "6b424a63e74ddf91bad2e4f268ff072f05c44e7a9d073158a264603943a842bf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.947265625, -0.9765625, -0.193359375, 0.86328125], "student_probs": [0.09794123470783234, 0.09511347860097885, 0.2081531137228012, 0.5987921357154846], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a40dca95a6eabada554a88d48aec5114014b0574bbaf010d9f1a635198284b4d:action", "state_id": "bea4b7289ddd0b0ec7139fd318fcbf191fb27ee81091b0b32fe26ab631376ec0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1171875, -1.2265625, -0.46728515625, 0.74609375], "student_probs": [0.09749821573495865, 0.08739684522151947, 0.1867435872554779, 0.6283613443374634], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "46c72207503c68acccd42ec7facf887a3bad466b7735ebc96757f2eccb6799d4:action", "state_id": "88bb1d80ba6492f9ffd22962c7ca48328c15f90d8c3a7221dc2253f4a196b958", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03125, -0.267578125, -0.68212890625, 0.9765625], "student_probs": [0.08325720578432083, 0.17868179082870483, 0.11804381012916565, 0.6200171709060669], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2249d6ad8f89c9002a9194996b260122cf3a72352476ebc1f3143499c002fd97:action", "state_id": "44f574a1b5dd7fa034c407868e517cd02a21371a87bbb47eae80ec0e982f1a13", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37890625, -1.3125, -0.64208984375, 0.541015625], "student_probs": [0.09108828008174896, 0.09734248369932175, 0.1903083473443985, 0.6212608814239502], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7239bf98901938ca4c3ee1b8301bce8f57cc2af1f37caae6f2d32a77da7911e0:action", "state_id": "026e07d0707a4f980932e2e997d99ff363fdfa504538d55dcf70eac9b2e282e8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.31640625, -1.34765625, -0.7939453125, 0.5546875], "student_probs": [0.0985143780708313, 0.09548341482877731, 0.16611221432685852, 0.6398900151252747], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5649f9cae1aae80616b57defd43d32476070b6722420274471d89c3acba6ae8c:action", "state_id": "8ca43f0e92bc4860cafb1002df44bcd49fe14146d790348244310f57db997d1e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.453125, -1.36328125, -0.67578125, 0.337890625], "student_probs": [0.09741625189781189, 0.10657370835542679, 0.21194709837436676, 0.5840628743171692], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "62cf03902b579c7a5ef1bccad33b20070bc09a66ed1419f2166fafca3d47f010:action", "state_id": "deeec24134556c49acf4f2714b7f6a91b6176b36819ad458ae382351bed24309", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.08203125, -0.994140625, -0.185546875, 0.728515625], "student_probs": [0.09383829683065414, 0.10245909541845322, 0.2299949675798416, 0.573707640171051], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7d6ae7a7dec2c92cacf5126c70f1bb609659a061b876215e6c62a3fd066fabf2:action", "state_id": "8f874c6e509e6f739a1185d19e923bcc20dd76bbe85c4e89767a7b457c664eec", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0546875, -0.98046875, -0.12890625, 0.74609375], "student_probs": [0.09385038912296295, 0.10108084231615067, 0.23686327040195465, 0.5682054758071899], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0b4a4b8e7aef00f784b3e840d975533ef3d07aa918c3d83c5c884d9e7dce9dfd:action", "state_id": "5d8fade80233b5ecf4d50f118eada950ed9c6904c3a37e4483ed95147aa6049f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -0.998046875, 0.00390625, 0.771484375], "student_probs": [0.09080322831869125, 0.09478997439146042, 0.25816959142684937, 0.5562372207641602], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "10ada4e33de70edf6940b8c67af31573eab7c720d3ded4940b40a671fbfbb3fb:action", "state_id": "77840b427a5c18c96487510f5247a1bc70a1773a973eae521350d5b61f7b8bf9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03125, -0.9765625, 0.0390625, 0.7890625], "student_probs": [0.08971597254276276, 0.09475894272327423, 0.2616378366947174, 0.5538873076438904], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6184ab3accf7d7dca6fc7b6eb68318b2ddfcbd531be2211b09afd2b226ecc7c9:action", "state_id": "8177e7d4c6785fa55e3a37643a49039725cd76b9fffb7e4a8a925c4eb0c522c7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -0.98828125, 0.03515625, 0.802734375], "student_probs": [0.08890823274850845, 0.09317503124475479, 0.25928226113319397, 0.5586344599723816], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "704c4a85e32caddcee72f245de807329210defa0fe3c1b21fe8ab8ca84ff0168:action", "state_id": "4ef20895104f0c4d751c10288b38e55019ae56522d1f989a6b80d69750040311", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.97265625, 0.01171875, 0.8046875], "student_probs": [0.08937457948923111, 0.09495310485363007, 0.25410768389701843, 0.5615646243095398], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "106d7b76ef02c07bfae80bb336644512c616bea6b601f440010c692a866a0779:action", "state_id": "a10320d2677c43be451f3a4e143507a64307293fe6b9f03806c93cf83fe5c021", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.02734375, -0.9765625, 0.0390625, 0.8203125], "student_probs": [0.08848035335540771, 0.0930895283818245, 0.2570284307003021, 0.5614017248153687], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f11cbce76ea73e6d55b8f4316f6f80180f1cd5e02878caaebc269af8fa4a6855:action", "state_id": "ebadc511ebb969d1cfebd3406e5163f142bf23a5b3e55dc58ab0666d5dd264bb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.982421875, 0.046875, 0.828125], "student_probs": [0.08812036365270615, 0.09198930859565735, 0.2574869990348816, 0.5624033212661743], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0eee3a1586a743995c4ca910c24b3adf4c572ffe708d4c3318e4a8c717688a3:action", "state_id": "58adc7863ea2f1b889bb3ef7889cc2e3bf368f7b12867109d1df95260ffca926", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -1.0, 0.033203125, 0.810546875], "student_probs": [0.08803164958953857, 0.09189670532941818, 0.25823456048965454, 0.5618371367454529], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a7efe5186d8fab34c76641fc332f67d6feb6213a5e0d5d1565244247ac83a49e:action", "state_id": "b050e552ebbe9a1a8d2a495b7f8ca11de68482b1cbecc4c0f7472f3819d28972", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.046875, -1.0390625, 0.025390625, 0.810546875], "student_probs": [0.08820650726556778, 0.08889831602573395, 0.25773870944976807, 0.5651564598083496], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "97013790d2e099d25077cab8870cdb837c42940edf211dc6607a507fc2cdc6b8:action", "state_id": "07259e39296092e64e78814a32707d9ec5d72a5bc4f5188c083d322dfc6e41fe", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01171875, -0.97265625, 0.076171875, 0.853515625], "student_probs": [0.08721975982189178, 0.09069420397281647, 0.25886884331703186, 0.5632171630859375], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "65f9f2857f37b42e1060427728357fca9109d5b44207313e6cd52cd5b9c0474b:action", "state_id": "e595b1839487b3f7d2392ea3fdf8f7e7113ad950c15461af02e271fc8fc5002f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.00390625, -0.982421875, 0.05859375, 0.822265625], "student_probs": [0.08988456428050995, 0.09183657169342041, 0.26008960604667664, 0.5581892728805542], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dee8da9a879b4430cd0a15e1386d1e7cde31119c070e15b790a3dd152ba99532:action", "state_id": "5ce15de56b536b1eb2984a2137a1c3ce0c846d3a8d456c964ee8b8ca25e308d9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -1.021484375, 0.025390625, 0.779296875], "student_probs": [0.09058435261249542, 0.09183131903409958, 0.26160308718681335, 0.5559812784194946], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bf3bcc1ca81166ecd33f2c826f85e73c94bc2dc2d2b9301448c00f86a8d8469c:action", "state_id": "5b24fecde0f43ea387c08892aef8fd6b6e15f0e95dc422603a563a144531e1a7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.02734375, -1.009765625, -0.001953125, 0.787109375], "student_probs": [0.09137730300426483, 0.09299773722887039, 0.2547767758369446, 0.5608481764793396], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bf9c8063a417cdacf495257971fa2395de5041494928ac65b2fa58c32218e5fe:action", "state_id": "157cefa53093c3bf75c5125cf2c265666e171332e44b095d1df5c779e8168dc0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.994140625, -0.025390625, 0.771484375], "student_probs": [0.09210216253995895, 0.09577109664678574, 0.2523232400417328, 0.5598035454750061], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "50abe8c5c0de8c08f99cf421a685b27184fdfdfc22c49c6266f9f96e780b1bfb:action", "state_id": "970e09f4433ee8c661a44eaed7cd9533fcbc7a9667b846c05c7cb2100bd2415a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.044921875, -1.099609375, -0.078125, 0.728515625], "student_probs": [0.09553508460521698, 0.09045080095529556, 0.2512103021144867, 0.5628038048744202], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dad1912cd85de2adb4b1461c91071af10ecd03c0524d7fca35a0fc67194ca311:action", "state_id": "0a18aeec2895f857059c0be2290a3115603ec38fbea1d0bc04114bbf77500dda", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.064453125, -1.1171875, -0.0546875, 0.744140625], "student_probs": [0.09263160824775696, 0.08787330985069275, 0.2542698383331299, 0.5652251839637756], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3f7b7861d331e34bbfbd3e104afc483b9ede1404aadf32da00af34aa637e5348:action", "state_id": "f4d5eb5b78e85626503dcdb9814cb9c2d7995a2b14bdc7ba8e3fae0575fec05a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.13671875, -1.43359375, -0.251953125, 0.568359375], "student_probs": [0.10344076156616211, 0.0768706426024437, 0.25057658553123474, 0.5691119432449341], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8ac982c30b1089e057b39c4c5ca7b19c7e4a7c54d74efae13db1051d2ed3a0eb:action", "state_id": "0d56710e9464d14a856d6d75b50d7630569d2919fab116f400514f256ff38d2a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.990234375, -1.05859375, -0.056640625, 0.8359375], "student_probs": [0.09356614202260971, 0.08738373219966888, 0.23799802362918854, 0.5810521245002747], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a5c6a98ea5fc831cdf842d2b266ded87442b346acaaaa6629e0f0c38fc0f6660:action", "state_id": "73b48cec7a69c113579c089ef9ffcd031e2025e3d14cb154d800a1f03e1052aa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.083984375, -1.11328125, -0.064453125, 0.732421875], "student_probs": [0.09180425107479095, 0.0891536995768547, 0.2544717788696289, 0.5645703077316284], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0c2420b3a0ebc0fd6e26b8949c778a4521bb4060613b4b70451847010e324ab:action", "state_id": "0346985157d5ca4440c1e3e419ea8422e79912391ffbfb113d2b2bf46a211569", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.111328125, -1.009765625, -0.02734375, 0.736328125], "student_probs": [0.08765573799610138, 0.09702605754137039, 0.25914856791496277, 0.5561696290969849], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "93d9a870c0b3d78f62859c30f707cfb87c9201693007b65eb3c34c4243348c5a:action", "state_id": "51b3a217afb614d203e46f7f5f31173ec2cc2492739dc4fa60684fe430cfcd2a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.111328125, -1.41796875, -0.296875, 0.517578125], "student_probs": [0.10998497158288956, 0.08093959838151932, 0.24833953380584717, 0.5607358813285828], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "40782d667576971b53e40229d94f3a261edb73f4a473f2158966859709d7e4f1:action", "state_id": "883fa2e2d1a465e8f0dcb540cbc894b888a305b562a3748ebb669ded6c05531b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9921875, -1.048828125, -0.091796875, 0.8046875], "student_probs": [0.09581967443227768, 0.090543232858181, 0.23577044904232025, 0.5778666138648987], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "04d233178386af8a7004a98a716f794ca8ba09ea56bd9d58bd033769804b74f4:action", "state_id": "d988db1159080a802ee12efb34b8cacce2da2dbf25d6211020042b1977990c1d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -1.0390625, -0.060546875, 0.75390625], "student_probs": [0.09340903162956238, 0.09377462416887283, 0.2494877725839615, 0.5633285641670227], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f38311c313876e1cb76a598987780957135e43d8d5c78a1fe855c89795b81ed1:action", "state_id": "61d66450e8a1f8e7a069db75e16e26a9818744867dd960417df92608609b2821", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.951171875, -0.0078125, 0.7890625], "student_probs": [0.09106433391571045, 0.09808014333248138, 0.2519282400608063, 0.5589272379875183], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0bad5f08232411cf176e87780a8e8dc06ea72581133cabc1186f425d7b8bc8d7:action", "state_id": "0b2155b954f26e1919160ad33073495e42066cfe0bc9c7bb4d21a3fb9455244b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.009765625, -0.927734375, 0.015625, 0.798828125], "student_probs": [0.09111173450946808, 0.09890085458755493, 0.2540363073348999, 0.5559511184692383], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "daa6af4926ed0c742f2771b6988cea619130d919499b210c101b2df3b48bada2:action", "state_id": "a3972c144831b97ac5b15b99974cef751cc7e6887a1c35b38590c70fe6bf169c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -0.92578125, -0.005859375, 0.814453125], "student_probs": [0.09030703455209732, 0.09879619628190994, 0.24788898229599, 0.5630078315734863], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7a1f3f8095a9d6de4948af614687b3e9ed122f66a466a3725800d7679d56c312:action", "state_id": "dff9033c4518b68e657ae7f1a1232520e538e8f81d4d495154923a34c9b5ec37", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.99609375, -0.91796875, 0.01171875, 0.80078125], "student_probs": [0.09215203672647476, 0.09964010864496231, 0.25245988368988037, 0.555747926235199], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "69777e61048821339399240f0af7993d2a81dd1ea594021e88c377c9189d5af9:action", "state_id": "6364d05632b92ec99b4f3b3315083e89927acbcc33948869dab4dda18e8ac4bf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0, -0.908203125, 0.0078125, 0.83984375], "student_probs": [0.08983678370714188, 0.09847388416528702, 0.2461169958114624, 0.5655723810195923], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "087ef9b7f6220f2c41c5d5064c23fc8809da2ff1bf90e6590fbf0b184093eea9:action", "state_id": "5058e7e5c8d8618148135d5ddbd9e2101c38853ab3f28193e334586bfbbc1996", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.994140625, -0.92578125, 0.033203125, 0.841796875], "student_probs": [0.08980372548103333, 0.09615733474493027, 0.25087884068489075, 0.5631600618362427], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "124fdbb7c7d5df6384bbba5ad5e02f0c809526bce94a0944cba2c695dc5f4204:action", "state_id": "f488208ed77bf80f74f2f6860066dbc6dd12e0d337c25b1e45c1e7970a40f10c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01953125, -0.943359375, 0.025390625, 0.830078125], "student_probs": [0.08865515142679214, 0.09567202627658844, 0.25206223130226135, 0.5636105537414551], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "87fe4bc8039f173feb0e559db86782ecea809863d88acf98cba8846775a1aed4:action", "state_id": "ec071ace1ef53ac6a262b49f832e0669b206a00c4e1c451acfccbd43e0c31f20", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -0.94140625, 0.01171875, 0.830078125], "student_probs": [0.08925998210906982, 0.09613677859306335, 0.2493598461151123, 0.5652433633804321], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "287dcd02e06ba608155fbfd266df20590a38dc2a6710a34714e3bdd2f9bfd02a:action", "state_id": "078a635832dc915c46f47a597b8e75c8b21163dcb708c6a7329ec1fd8ed6d2cc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0234375, -0.95703125, 0.029296875, 0.798828125], "student_probs": [0.08992810547351837, 0.09610264003276825, 0.2576868236064911, 0.5562824606895447], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bd90631c45000b77841081411e5c891882429346f0514b357551d46e36c6d575:action", "state_id": "aebf487b0f7341f25bb442255c9f0313168ded795719c9b16de4a55689774f11", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.953125, 0.04296875, 0.822265625], "student_probs": [0.08763092756271362, 0.09493687003850937, 0.2570590674877167, 0.560373067855835], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81ea9adaedc860c833030d982551d96421c1ab0b297dba454db76d026acb4029:action", "state_id": "410980259c52c5b616511a47f66131293c4e5a8b5e7dac7d657e73a18d701535", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -0.953125, 0.005859375, 0.796875], "student_probs": [0.08909579366445541, 0.09728090465068817, 0.2538102865219116, 0.5598129630088806], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8b23c19d3da4f75647956dc97cde36f04e12cb422ff50e1d5d8172904ff9517c:action", "state_id": "d262a0ad203b541c1c2007d21320f9fcc8d67cfce7c96c9d581d6c40e0b5cc42", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.958984375, 0.033203125, 0.7890625], "student_probs": [0.08954298496246338, 0.0964415892958641, 0.26011529564857483, 0.5539001226425171], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81273e14bbddfa205435e9cf0730d5851787dedb8ed995647bf492b2d6b46af3:action", "state_id": "a605ebe9c229b84b946e42d3a052b7e148f2f783eac505ce4318ac32a7f83a04", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.037109375, -0.9765625, 0.033203125, 0.796875], "student_probs": [0.08898790180683136, 0.09454229474067688, 0.25951460003852844, 0.5569552183151245], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6b1591131e2550f279005862f86a08cfc9d0f91222ba43e5821e42c56fb7f58b:action", "state_id": "53c2c73bfbb33f72e1d065fbe236ff5da967d6dd6c41fd84c1fd566982903648", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.97265625, 0.048828125, 0.7890625], "student_probs": [0.08929415047168732, 0.09486766159534454, 0.2634773254394531, 0.5523608922958374], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8a7205f09bee5863cdcc56c429e1db16d436b9d65d4c0b850736924d86ff856f:action", "state_id": "9b7b29856e029fc1509be067a6e85fd80d29e2c8c1be58024ba82ef7477c83b1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -0.970703125, 0.05859375, 0.78125], "student_probs": [0.08863868564367294, 0.09528134018182755, 0.26670169830322266, 0.5493782162666321], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0270f682cb3f60c46d1d0df1aeb6051ac0ee95d84253e4396dce954053e013a7:action", "state_id": "5d4b6bc563a702e2c68862142368122de05298a379cd6b7229ea101eb78afa55", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -1.005859375, 0.046875, 0.763671875], "student_probs": [0.0902375653386116, 0.09346640110015869, 0.2678256928920746, 0.5484703779220581], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "512f745dc67a50e7e0c8d146746b5e79a385d411169ff5062a0a6971feba1489:action", "state_id": "62b653b62e748e0eb791bb4176b115ccc2e39bfe504f62d9285d8181c8803042", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -0.994140625, 0.025390625, 0.763671875], "student_probs": [0.09065373986959457, 0.09500430524349213, 0.2633419632911682, 0.5509999394416809], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "65794ed41c5d497d9ca26f7d55cd3426fb345e6d366bcf5725f36826b7d48a43:action", "state_id": "d2ccdbdf928476528c53266d1a1b528fadb7ece3f498a06e1bb6baaf9229f1a9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -1.005859375, 0.017578125, 0.771484375], "student_probs": [0.09070919454097748, 0.09377158433198929, 0.260942280292511, 0.5545769333839417], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81fec5eba5a11851dd735f5d519d716254181a82608fa0fca01462d9a849b492:action", "state_id": "e2d0548c63d90d105adf41a0ffc95d5ed8d3286b5c1aefa6511e116c226e56d3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.05078125, -0.986328125, -0.052734375, 0.69140625], "student_probs": [0.09533457458019257, 0.10168153792619705, 0.2586406171321869, 0.5443432927131653], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3440fa2b4d0073459a247fa5d04d550f8540b65f878d43c8c62cc18914b49157:action", "state_id": "c87a557a029d8d385f334878baf1fd8b4bb26a786a373dd1447cb02e2d5f6c82", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.08984375, -0.955078125, -0.041015625, 0.765625], "student_probs": [0.08777112513780594, 0.10043374449014664, 0.25052550435066223, 0.5612695813179016], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "34fed9543ee55a805c70b69b738af69c2777f75124752af2d75fe8e5b5cf3b26:action", "state_id": "89dc29162b95b64538d13c3eee2f1677cf4f31d1a48dde4c8ccd8296df248eae", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.65234375, -1.77734375, -1.53125, 0.470703125], "student_probs": [0.08796786516904831, 0.07763136178255081, 0.09929203242063522, 0.7351087331771851], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a4e858c5863eda1170d10b2a3162b9bb4e22804975da569008961497a2b6ac0e:action", "state_id": "735c2d682b4907550471749d1efc3664165acce467fdfe30360e7c710a43ceee", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55078125, -1.58203125, -1.2734375, 0.5], "student_probs": [0.0903928279876709, 0.08761173486709595, 0.11928417533636093, 0.7027113437652588], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7c40c98f88473717fe6d47b4892df162315c48bf04ff1b8669541a99cdb114c5:action", "state_id": "2d6d330d5d9bc588fdfc1b907119cd763f6adce29266f4078fda508d7bb266a5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.54296875, -1.54296875, -1.2734375, 0.5078125], "student_probs": [0.09022565186023712, 0.09022565186023712, 0.11813700944185257, 0.7014117240905762], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "35a9e90740f4f397839d9f7200bcb82e74c686a44730e1439140b7246f504253:action", "state_id": "b77b1e15ec234cd3a0b1e44b0c774dcffd85c0d6baeedafab044b2fbb7a00160", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55078125, -1.390625, -1.2421875, 0.46875], "student_probs": [0.09033428132534027, 0.10602482408285141, 0.12299094349145889, 0.680649995803833], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "78304dabffc84efc50301e619dd56a4968e7ddfbec12ad0f27e06d05c3363380:action", "state_id": "39c66e989ea2023a6721a9a36fb08df7d46a4edc50afbda427143bad8ea8b677", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.12109375, -1.4609375, -0.31640625, 0.46484375], "student_probs": [0.11322788894176483, 0.08060483634471893, 0.253177285194397, 0.5529900193214417], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "01f09ad6539595b98ee1c3305552c34be903ab5a910e4e2ef6bbe055d06536ee:action", "state_id": "bc63555edb052cbc2fa71123c1e634e359f64b6312481ba3c82e0309e2efb0ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.978515625, -1.115234375, -0.1640625, 0.748046875], "student_probs": [0.10254881531000137, 0.08944465965032578, 0.23154912889003754, 0.5764573812484741], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3c19fb6eea194e2d511f739e345594bd7329ca90ab01b37552a5c2a9dcdc3efd:action", "state_id": "5d05e6bde175d31a55a7ad38f5aad7fd556ebc774371c434e03e69472bb421f2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.068359375, -1.12109375, -0.15625, 0.693359375], "student_probs": [0.0974598377943039, 0.09245351701974869, 0.2426329255104065, 0.5674536824226379], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b22a00abc61db73ccc656351c9c8e00cbec7a3a557c0a106499e201b5dec78d9:action", "state_id": "d85af456736f6e23bea859ec12676fb83081f62484516fa02c4bd84b9991c88f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.009765625, -1.01171875, -0.025390625, 0.7578125], "student_probs": [0.09495918452739716, 0.09477390348911285, 0.2541239559650421, 0.5561429262161255], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a5a0022d4d1c092383c613a754745f68f537f6cb5b0ad57d0ee9c83f45a6ad8a:action", "state_id": "2749ddcfccf9fd1158e9cc853a9af395679aca6a227fdf06cc2354de8fa630e8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.25, -1.37890625, -0.4935302734375, 1.1875], "student_probs": [0.06470736116170883, 0.05688142031431198, 0.13787461817264557, 0.7405365705490112], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a8f570f9814287f27cc9bc7d7929d643c18a340536c9b4649439cf66c08dfed4:action", "state_id": "4b20ac195e0b7730e3909d2f2b964f6e44ad1f92952d03c70bb629dc491510ad", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4375, -1.46484375, -1.11328125, 0.89453125], "student_probs": [0.07323423773050308, 0.07125886529684067, 0.10127926617860794, 0.7542276382446289], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d33c0841f8a10acc25981edc76e5accd1e52fb78d6f9073c553a0a213235460d:action", "state_id": "a8e8f21dd1047772417b1e3534f601d82b815bde49413c3bcbe7f931521ddcfb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.50390625, -1.515625, -1.24609375, 0.787109375], "student_probs": [0.07594504207372665, 0.07506024837493896, 0.0982801765203476, 0.7507145404815674], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "16cd8383344ced68491c609c42a55b247f8afe5f8f9b9c48e2dbf2a152b50bb9:action", "state_id": "93ceb617af3fb5afd6ef0098339c4b21bae6c5e3a5698ca7849ce09844febe02", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.38671875, -1.34375, -0.935546875, 0.830078125], "student_probs": [0.07817421108484268, 0.08160647004842758, 0.1227453276515007, 0.7174739837646484], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "502da2b6e2127aac6e00ff99449a9847dbaa9b895a20adf444318ac926d7016e:action", "state_id": "0989399fd2756a52493f921e11c2f18df5b291de8fc339ddb10e3faf80d4d047", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.52734375, -1.66796875, -1.375, 0.763671875], "student_probs": [0.0774092823266983, 0.06725434958934784, 0.09014778584241867, 0.7651885747909546], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "42dd9754f61acc5c62eba52899c7a99685f47ed47a0aad7c970b37397cd2402e:action", "state_id": "3527ce27d6e99394db6a12abc4b943e5dd64ad3af98318f2e99eabd7d9d92e5a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.359375, -1.328125, -1.029296875, 0.884765625], "student_probs": [0.07778997719287872, 0.0802592933177948, 0.10821183025836945, 0.733738899230957], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2838858f278aa009e521ac8a2fc1e53a5d75aa7a924599d273420ff2c9f07ae1:action", "state_id": "4be8bb067481f50a73289f6762c86570ed058ecdb14ee6ce2d9144f0c2e4d9b8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.34375, -1.421875, -0.958984375, 0.859375], "student_probs": [0.08033820241689682, 0.07430069148540497, 0.11803850531578064, 0.7273226976394653], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0c5882d128b74950ed396b551e442933dede1dd45b8621178d571e3ac1770341:action", "state_id": "c8d6e796c09953597a42a73a2100b5e874e769aeffa254e3b66ebc796fe2870b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.359375, -1.34765625, -0.962890625, 0.767578125], "student_probs": [0.0841209664940834, 0.0851125568151474, 0.1250533163547516, 0.7057132124900818], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bf568f2291f6f32b8d014602d5379ff6572820e4227b590e554559cfcc7e58c8:action", "state_id": "b9fe075b4694b98cd83ee8900dfaa5b3b070d2d9dc27a2a575c4be6060ef8a6d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3046875, -1.27734375, -0.8984375, 0.833984375], "student_probs": [0.08321505784988403, 0.08552186191082001, 0.12492059171199799, 0.7063424587249756], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3775dcde2fea3c63239f877f5a5b8ec24f1655d82867eb50b3f4df8788c611aa:action", "state_id": "e1fb7594cac2c9f58d75b9bdc06b23b001a020094fb11c2272dd72ff68351682", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.09765625, -1.091796875, -0.596923828125, 0.83203125], "student_probs": [0.09484819322824478, 0.09540557861328125, 0.15649281442165375, 0.653253436088562], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "95e9d3d1b6738cd7f1bf318f58f3126094e484555e76e356a23b800bb7985828:action", "state_id": "06d3afd8ea097ce8ce493d89f928c0294fd53d036c7a390bab9f14b34d44bc1e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -1.00390625, -0.5318603515625, 0.91015625], "student_probs": [0.09312182664871216, 0.09664243459701538, 0.1549440175294876, 0.6552916765213013], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e65a668b8b39ced439d87479c795c085ab7e47a466ab2fd12610b93b1a676b7b:action", "state_id": "0a8f9b43bb7972a518206659828d0958ca1e9e8b6fea16b32d0ccd6c7b359656", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -0.984375, -0.43701171875, 0.919921875], "student_probs": [0.09144692867994308, 0.09621064364910126, 0.16631828248500824, 0.6460241675376892], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f3ea3efeae61659c95b91c3c42ba2aebb1b8d796d2b64037d961476244912832:action", "state_id": "98feaa8f46c7b03d26700bcc023dbaa05e0a618f0661fcf6e87870604178664e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0703125, -1.00390625, -0.2890625, 0.810546875], "student_probs": [0.0924898087978363, 0.09884023666381836, 0.20201632380485535, 0.6066535711288452], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d0fdd38979939dad9c1ff188576a0aa6804f690b34ba01bc8cc35f47ccb5401e:action", "state_id": "4c4e4a14b03237bbaeb6b4d7d9e4cf13114d617cabe1adeda7b3519ba89a5948", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.130859375, -1.013671875, -0.41650390625, 0.83203125], "student_probs": [0.0885968804359436, 0.09961216151714325, 0.18099187314510345, 0.6307990550994873], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1200c9bf6d93e0ea60c8d63244a115e1c4cfb85bb0ef0413b4461ddc9b1d2ef2:action", "state_id": "18ff6c3f0b475352b979c787c4aee7b13e65b004b5b8d0518ab1f2699f44dfaa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.095703125, -1.076171875, -0.51611328125, 0.84765625], "student_probs": [0.09270250052213669, 0.09453088790178299, 0.16550232470035553, 0.6472643613815308], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aee26c1fc701fa05bb986773d88073e2d3ed8872e250e3ab3e98de5b174e792b:action", "state_id": "559a1b8a1c1822a0a1b86d7af3cb348b29b937bf49bba464dd5fe0f902aaeb62", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.126953125, -1.052734375, -0.47900390625, 0.759765625], "student_probs": [0.09446131438016891, 0.10173884779214859, 0.18057380616664886, 0.62322598695755], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "33d68def4ae6e361b967138fc3671897cfbe5186082b3765c88527e63bd25eaf:action", "state_id": "8313bf47872430601c756d28242d96447b3df70071dd993f60dc006ea42c716e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2265625, -1.25390625, -0.7119140625, 0.859375], "student_probs": [0.0854831337928772, 0.08317737281322479, 0.14301757514476776, 0.6883218884468079], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "45038a852428957436dc806fb9567869c880a5d871a9ddccec9f933cf9465b59:action", "state_id": "57b5f12b0c92637be4a47bd4ece7675c9f9d917b51f14d9a910f8b147aa65c45", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.546875, -1.59765625, -1.015625, 0.451171875], "student_probs": [0.09069321304559708, 0.08620268106460571, 0.1542743593454361, 0.6688296794891357], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "69e02d9a0aeda9eea5ebaa3ecf3f6a094e777bad4db2a5db6f9ef2a1953e389a:action", "state_id": "73e2a3878a0758d90d631053be35a8887b06302e7a52319545af0d424411d29e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.998046875, -1.013671875, -0.279296875, 1.06640625], "student_probs": [0.08391127735376358, 0.0826103463768959, 0.17217475175857544, 0.6613036394119263], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4dd8be92ce195bfc937ccef18707a4a5eeeb4cf392efa7eaae23e19465ded6c3:action", "state_id": "74350fb2d7fc3c863a81b72f81f3cede8d98e3c30a56ad376b745d3ae563dbf4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -1.060546875, -0.2578125, 0.982421875], "student_probs": [0.08722719550132751, 0.08339549601078033, 0.1861083060503006, 0.6432690620422363], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5113754095b81b02d42198e54c62bdfc7fcdadefa28fb31a110f3f6d0f0dfd4b:action", "state_id": "3f456340d70c218b4e7e12b97aab639118e8010b907812cc10d3aecba6dfab1a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3671875, -1.33203125, -0.9296875, 0.740234375], "student_probs": [0.08466242253780365, 0.0876917690038681, 0.13112773001194, 0.696518063545227], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ab429d1c55900101e1d972108d95e4b763f2c809d87e3770769097eb000b8893:action", "state_id": "72640df1dff0ce961e69d4dbf5cf6a900a431915a697c80bcb5a67b973543676", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.296875, -1.3203125, -0.6806640625, 0.837890625], "student_probs": [0.0814080461859703, 0.07952223718166351, 0.15075938403606415, 0.6883103251457214], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "27520aa042ae625da3e7c6d57761e1222328495102438e267f2889a57c9f6491:action", "state_id": "ddb7135f7a6e9f1a490226279f6280aadcd0728c85bbaa46d44165c5f2078275", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5078125, -1.375, -1.01953125, 0.408203125], "student_probs": [0.09464871138334274, 0.10809221118688583, 0.15423130989074707, 0.6430277228355408], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5a23478b2cfdaab338f27aaccf32e1d4d79b4ef5b2d0ac458a5dbfba9b8ceab7:action", "state_id": "f8ddb51b98865055d71fca34836e1b0dd3e5d82bab0036255a6223fd735c19bb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.62109375, -1.45703125, -1.18359375, 0.3828125], "student_probs": [0.08972557634115219, 0.10572256147861481, 0.13896968960762024, 0.6655821800231934], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "794262b15624a662131c3b6d5e46eb6f99e0ab44f5b6b41cd1c1bffe3419f28f:action", "state_id": "af6c549a838424f1ee0c0049f75843499c445eee0a617c2a6f5c25a683b5c9c6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1796875, -1.05859375, -0.2900390625, 0.76171875], "student_probs": [0.08671862632036209, 0.097881980240345, 0.2110968679189682, 0.6043025255203247], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5b855df64c47db872921768198b1f667af2442c23865abf93a79eae6ff66ee5a:action", "state_id": "de19992bf04517c25bdb611eeb88f3f60075560f554010b48b2e2c5108b00137", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.078125, -0.98828125, -0.154296875, 0.84375], "student_probs": [0.08736160397529602, 0.09557387977838516, 0.22005639970302582, 0.5970081686973572], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6c18889b7864ee76a7467b5edc0f96fe9dbb0ca2a43ff585c80ceccc613101f5:action", "state_id": "f24a0a448d8f078e48b9e68a105d1e94e08f3434fee0fcca415e4e660551ed29", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0859375, -0.9921875, -0.14453125, 0.826171875], "student_probs": [0.08749505132436752, 0.096094511449337, 0.22430090606212616, 0.5921095013618469], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "de9886990bb09c17ccb63e3036be453153f5462983e246cd97cbdd874315e815:action", "state_id": "1c7b37b1ae4ad20c19e6bd4ff498c406cafe8cfaf64bfb35b83a88e7dbbf09e5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.080078125, -0.986328125, -0.09765625, 0.826171875], "student_probs": [0.08697910606861115, 0.09552786499261856, 0.23231397569179535, 0.5851790904998779], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "86ab791785a2b766676bd30afb9f9797e82dcc2dcc0b700bf5ffc020d46abe09:action", "state_id": "1a912a3d5daa8aac282a84dd4bcdd3bf4d8e6f5c9faaac94cdc523b3c5970c7c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1015625, -1.017578125, -0.1328125, 0.7578125], "student_probs": [0.08975062519311905, 0.0976138487458229, 0.23646138608455658, 0.5761741399765015], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "35404ce70cddadbcc37c7f3aef4fa32ba0fcda5a31008b9af647a76609094240:action", "state_id": "bdf1a60a5684588425a41dae44766de4c081090932690b6365890c65a17a0521", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.078125, -0.953125, -0.236328125, 0.8671875], "student_probs": [0.08734007179737091, 0.09896926581859589, 0.20267550647258759, 0.6110152006149292], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "30829a2c2260c3491f784575c5d827f9ccf6b6f434355a0c53997514530bf8cf:action", "state_id": "e331312b74a82933451a48d957224aa23b9987d710f4d8039ca67e592b0b756e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0859375, -0.962890625, -0.2919921875, 0.859375], "student_probs": [0.08819227665662766, 0.09973994642496109, 0.19509072601795197, 0.6169770956039429], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f1c37a5dbefda3e207769de68b69d5fe0f268c706ae3bdbe832098982d97be8f:action", "state_id": "f3ef0878ad14ff4cab1168a8f366aa6ff4d15080434dfd73ec06ec6ebcba68ee", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.095703125, -0.951171875, -0.3232421875, 0.8515625], "student_probs": [0.08826062828302383, 0.10198497027158737, 0.19109202921390533, 0.618662416934967], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e544e05b6f88fd214b640b4f7950e5dd4b6b858a32c95aa4e950c011f7e764ae:action", "state_id": "56ce18934c2e1bf2ad4120580148a4a549c22b050a3a03ebc6b8576e355bb041", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.10546875, -0.978515625, -0.3642578125, 0.857421875], "student_probs": [0.0880768746137619, 0.09999930113554001, 0.18482713401317596, 0.6270966529846191], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bb518ea1ad9a9d181e34122f02ad76f1ae5461c94c4baa9bd0b42e668dd4b7be:action", "state_id": "fac1ce878bcdb024897bffc689bd7e2deddf66f50386bf8825610e51863aff52", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.123046875, -0.986328125, -0.41064453125, 0.806640625], "student_probs": [0.09031183272600174, 0.10354302078485489, 0.184135302901268, 0.6220098733901978], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "022a900c559388750c7da0df5adab7bec2f03af94d10a7bf5f286263befcc738:action", "state_id": "75d6553de4edf69407f6328d3c37163cbfa858f16af9bcb5e8b6f2cd111d6a11", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.171875, -1.017578125, -0.58984375, 0.859375], "student_probs": [0.08635491877794266, 0.10076213628053665, 0.1545468121767044, 0.6583362221717834], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "94f0cb6cfee218ed041e081b838d9b0db367bc17122e81f759a686c03a5bf2e8:action", "state_id": "3d63f74c3c375d9405c3f63dcc02f4d614eb146b3f0df873b4cfc7c4640b7e66", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5390625, -1.34375, -1.046875, 0.603515625], "student_probs": [0.08082140237092972, 0.09825383126735687, 0.1322149932384491, 0.6887097358703613], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ac8b8f40a5af8fa28a2831db5d4e77bc3e73c9c66322d2559b4ab88d892ccd07:action", "state_id": "bf9da1522197044c16444a7175961773d60b52aa338c8a448a3e2df3da5b8753", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.59765625, -1.4140625, -1.13671875, 0.51953125], "student_probs": [0.08267997950315475, 0.09934227913618088, 0.1310940533876419, 0.6868836879730225], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9c6c90877885f595a04285e303a4dddf3f2ca3b7d7e718ad84736b72a0930b9c:action", "state_id": "5e16e47f9e24129456f6f07f98b203723ff1ad27ab1a7efd8ec892438d694a20", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.515625, -1.31640625, -1.09765625, 0.7421875], "student_probs": [0.07517965137958527, 0.09175293147563934, 0.11418835818767548, 0.7188790440559387], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "281188e0a8b1e0367b3717e131f73896911db1adbd50fedede0f3149d57acfab:action", "state_id": "a7e05629bff780a403cff7d1926f099687e2deeaec3e2e3af74b2cccff942163", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4296875, -1.390625, -1.076171875, 0.95703125], "student_probs": [0.06972701847553253, 0.07250462472438812, 0.09929581731557846, 0.7584725618362427], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "440fc185b5f42f6ee9a328dbb488fbdfed7943b36bcf86cbd8c4c13823607214:action", "state_id": "b74b67f54efc71ceb05cbc8310e8f08358926fe7467e3c4543a7521a28f5e577", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.453125, -1.35546875, -1.10546875, 0.8359375], "student_probs": [0.0747160091996193, 0.08238064497709274, 0.10577885061502457, 0.737124502658844], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d2a7dfddfcd207afc75ffa8b69143575b10ef99d003b79e7b23bb9289cc4e9f8:action", "state_id": "faad4d0980eb7b85aecde02f09f2fc6610b9334325727b280876e9367c3c8419", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.49609375, -1.36328125, -1.16015625, 0.72265625], "student_probs": [0.07851070910692215, 0.08966204524040222, 0.10985623300075531, 0.7219710350036621], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "79ae0d376c5034b4aa77fecfb7ee5c7ed904b7448b8593ebccb7acd752dcfea3:action", "state_id": "53095abb6eff771f0379d38000e74740f92013b4786031d08f6d0496a55a8ce6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.50390625, -1.66015625, -1.40625, 0.66796875], "student_probs": [0.08523222804069519, 0.07290298491716385, 0.09397567063570023, 0.7478890419006348], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3c81917e8451195b38a2bf32fefcfc37abd1c8c49050d77567881fa12c5cc945:action", "state_id": "9d563105bb84cc77ad4766106789931cfff8e290c29e7ee164bd98c3ace3d272", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4296875, -1.390625, -1.15625, 0.54296875], "student_probs": [0.09484013170003891, 0.09861813485622406, 0.12466500699520111, 0.6818767189979553], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "66253e4045a40b5318cbfc7ea15289c9bb277996ed0cdf90cfc7ee6fd02cc240:action", "state_id": "75e35ba8af88f317847b7e7bc8ccacaef89f8d4cfe9804752fbcf3deec788f65", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.30078125, -1.203125, -0.8076171875, 1.076171875], "student_probs": [0.06890799850225449, 0.07597683370113373, 0.11283610016107559, 0.7422791123390198], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3d4b9cd65fd4f03252daa2d9c89e780c081f59902394b7114d0547e2dc6b897a:action", "state_id": "fe8263297a8da17977a5ae254fdd7b0c4173a3d2919b479b0b8de4f76ec5f420", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.171875, -1.0078125, -0.7333984375, 1.060546875], "student_probs": [0.07662219554185867, 0.09028301388025284, 0.11879073083400726, 0.7143040299415588], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d73e0e0cc7f8194591e3ab2aea0b5d9a1e9e3f50241a44ed498784271a86d9b2:action", "state_id": "6c8322f4ed0903a7e9a2952ab03c2e8727a63bf2103892222d6970f066cb7d14", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.16015625, -1.12109375, -0.57568359375, 1.150390625], "student_probs": [0.07187100499868393, 0.07473401725292206, 0.12893979251384735, 0.7244551181793213], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7ec3684d33dccbe74fa73a035c1d8ada3b059433d6ae9cf72001fc28cfdf9f96:action", "state_id": "bcea24b03b10afc894a59c2408909c70854d512df302d66cc255597bd7c8bcb8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.986328125, -1.060546875, -0.05859375, 1.146484375], "student_probs": [0.07754456996917725, 0.07199769467115402, 0.1960926502943039, 0.654365062713623], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6fbb9174938bada1c0227c6f7c0e94cd5d8e3cfe7ad5c54c8d986ec546d41a20:action", "state_id": "6daffb585e99e41bfec5845614310942a4ad45f40cde941357b7a0698f408df3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2578125, -1.359375, -0.783203125, 0.833984375], "student_probs": [0.08613066375255585, 0.0778125673532486, 0.13844524323940277, 0.6976115107536316], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "091269d07a4b32b5a5331d23c0a72ff334ecca408640f28685f947bd8f051186:action", "state_id": "bdce7b8b9d725c1117062186cc4cc62fbc2cc3b98bfd0187fd67193222aea9ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1484375, -1.1640625, -0.7744140625, 1.10546875], "student_probs": [0.07714300602674484, 0.07594701647758484, 0.11213285475969315, 0.7347770929336548], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e13d3f36e1ff95f2f87f0f602fd85aec5f1286e303b1b9901ed90edad4a67be8:action", "state_id": "ec24640c03ea8fc12143a876ef83c71754387421e00ab915f5cf855bd63d93ac", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.484375, -1.3828125, -1.14453125, 0.75], "student_probs": [0.07780726999044418, 0.08612480759620667, 0.10929806530475616, 0.7267699241638184], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4436727e29f6494e9bd1deb5ec3971aa100aa50432661e0a5aa9544a7a23a5cf:action", "state_id": "3d11d0dff6e5e68675949c2b331c2f693bf63944c6d3c460a9fc5074ec76ff05", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.58203125, -1.41796875, -1.25, 0.509765625], "student_probs": [0.08567849546670914, 0.10095395147800446, 0.11941839009523392, 0.6939492225646973], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fbb88dc29db24bb116461a8a082a86816ef9aa779ff8d7e1d5f9c0787f2737d4:action", "state_id": "57233f8351b89197a50e1c21a736040578f8ea57f3a25ea083e562d37ab9cf18", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.23828125, -1.37890625, -0.480712890625, 1.1875], "student_probs": [0.06530416011810303, 0.05673723667860031, 0.1392991989850998, 0.7386594414710999], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "52cefa4c9a06aa43a40e07db1576f3d7696f794da6712e955b9de6c3376fc1d7:action", "state_id": "894b1087d9cbe126c643104678efa32e29b22dce2baeff913fb1bfa816dcb1e5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.28125, -1.171875, -0.37890625, 1.111328125], "student_probs": [0.064422108232975, 0.07186806201934814, 0.158824622631073, 0.7048852443695068], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c0547490121179ad360405241dadb8aa2d023b23a28e991aef0dff63b5db5df4:action", "state_id": "5f41befdfc36e1faad0d1bd0daf003ff98484060e4d730f8d1e8946839a1b21e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3046875, -1.3359375, -0.44482421875, 0.71875], "student_probs": [0.08405936509370804, 0.08147313445806503, 0.19861863553524017, 0.6358488202095032], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "24919e511d9e0a822ab404ece88155dfe2544d7b844b48c7cc0f1f97f68b9c4a:action", "state_id": "b71cb7a16f8bd00df20c999d3c8875e6b945d3050e1eb9eebc05b2de11f0dc90", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.296875, -1.1328125, -0.359375, 0.564453125], "student_probs": [0.08957241475582123, 0.10554209351539612, 0.228731170296669, 0.5761542916297913], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7befd58dde3d8829cc31c85bdf45f3dced2e9ef44bdf148290a7bdc1eb8eeaae:action", "state_id": "c1af9a0ff2e036f8270f302e8ca3b5bbd35b07d43887b44b48c5f53412f37615", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.078125, -0.921875, -0.173828125, 0.78125], "student_probs": [0.09042379260063171, 0.10571612417697906, 0.22336435317993164, 0.5804957151412964], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5750b9712f17b7c5513f1da2486093033e302afd5286252754df8c3a1ac69672:action", "state_id": "cf666e730701cf3e898d85afe4e317ac023426710dceebae5d62bbc52976c36f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.998046875, -0.986328125, -0.095703125, 0.7421875], "student_probs": [0.09827197343111038, 0.0994303748011589, 0.24227723479270935, 0.5600204467773438], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9fd19e97ec6ad10e59644501279d96834a052c74d6fc0371edf13f570b21d6ac:action", "state_id": "07481b9c47b85919976d3c71849b96692e8b0339b537d80f5dd6281cbe4a9e13", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.17578125, -1.2890625, -0.541595458984375, 0.783203125], "student_probs": [0.09199203550815582, 0.08213964104652405, 0.17344972491264343, 0.6524186134338379], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cab545c6a2049cddb62d966ae25b751ac675b0e03284a023e78e1d22d4d03b81:action", "state_id": "8342888de1c4e9f81ace300eee5738ef1839766e712faa5cd3a1410cdb5e88a6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.99609375, -0.98046875, -0.35546875, 1.009765625], "student_probs": [0.08813762664794922, 0.08952558785676956, 0.16725580394268036, 0.6550809741020203], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d43fca306268cdc990668e681d5b54e8209bfacb2626a4a98047f27ec3ca05ed:action", "state_id": "1706f0363f2dc050a153eecbd0ce529b03e61ab2b73ec7b7979d4f36f44b6a33", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1015625, -1.0703125, -0.453125, 0.85546875], "student_probs": [0.09072290360927582, 0.0936027467250824, 0.17351208627223969, 0.6421622633934021], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c2338e37e94de2dc86ad87f825813fd0fabd0883e092f1a515f15100eca3f61d:action", "state_id": "063d83c36cd3a76d07a425f94721e1eeff40d6737b6bbfa3467757c87a6786e1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.109375, -1.009765625, -0.57330322265625, 0.9609375], "student_probs": [0.08516931533813477, 0.09408988058567047, 0.14557813107967377, 0.675162672996521], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e6e00c6a0373c11106937b2f250049d898508d9b40c0d9c19776ddc88858b99a:action", "state_id": "718d8955f216e54de3c67ed262008bd4685660a73a81eb9e2e64a1f05dcb2531", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.064453125, -0.984375, -0.5543212890625, 0.91015625], "student_probs": [0.09130178391933441, 0.0989137664437294, 0.15206411480903625, 0.6577202677726746], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "fac70a936b4cbadad0b820efbfcfb2053f09a1fa74e5dcacca2181d66aeaf37e:action", "state_id": "0543f9c87fdf2b6b7ce3892b10955fd0845cc022dbb82b4aa1e0334a4a8e9538", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.51171875, -1.296875, -1.1015625, 0.599609375], "student_probs": [0.08329220861196518, 0.10325470566749573, 0.1255258023738861, 0.6879273056983948], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "70d5a6493420536edb042e8eb1ae77b1d724eef981060eec4574668c7f8e4e82:action", "state_id": "b3eee8c93fdf19d60d4cf117eafbcba403cca2bdfd81f37febee7b9d880d99ef", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.46484375, -1.26953125, -1.11328125, 0.740234375], "student_probs": [0.07869097590446472, 0.0956638902425766, 0.11184242367744446, 0.713802695274353], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "af98c5e9c7045e9663acbb0b73a1f875949413894de23663fa89e821f940f6ac:action", "state_id": "e32636dbdeea1d586024cb5878c7772cabaa1e0b03ff218d3ee01988e29c2bb5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.099609375, -1.3984375, -0.203125, 0.578125], "student_probs": [0.10475513339042664, 0.07769550383090973, 0.25675180554389954, 0.5607975125312805], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "91223634a648920103ec5d929d041cc6454bca5d6ef22e15c510bc2f6c4cb91e:action", "state_id": "e16f5531c27baa6df2adc9b90359d00ea2aa44d9076b770d6d030ee04b065803", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9765625, -1.0390625, -0.037109375, 0.845703125], "student_probs": [0.09359920024871826, 0.08792831003665924, 0.23948119580745697, 0.5789912939071655], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "31cf12f60d07e32a2708f299621d3542062105bd4c6b4a4a7df6b6f0a721be96:action", "state_id": "22eca9dfb8c7c162c0d20e8c030f665879a098c03efcceeb78d290acd93abfd9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.048828125, -1.0390625, -0.03515625, 0.78515625], "student_probs": [0.09070879966020584, 0.09159896522760391, 0.2499663233757019, 0.5677258968353271], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3e0a9104c05803c67e0171e98ad3985279cf4addec4fbeb625b7d683219946a6:action", "state_id": "1e9aecd379acc1e914c4394ccc0b76956307b8bc81d20a28325f542977b67711", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.951171875, 0.01171875, 0.8125], "student_probs": [0.0888003557920456, 0.09639187157154083, 0.2524750828742981, 0.5623327493667603], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "59cdc549d0b13a82accbd441d0122767873646a44410ccb7d27795cf004e0127:action", "state_id": "f3262f3db92183ef98116f75160772d734a99cc39571ced61ef3596260c9b176", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.013671875, -0.923828125, 0.029296875, 0.83203125], "student_probs": [0.08877696841955185, 0.09712229669094086, 0.2519160509109497, 0.5621846318244934], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7ac54d0e9e0c67fb564c6f9d363413dbe1a7df96ccf6ce96bb4e870dddfda62c:action", "state_id": "5b956f3e2c8bedce5acac1d26e7c070b38008ef8605eb2a6520cacd1f4cd369e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.001953125, -0.923828125, 0.029296875, 0.814453125], "student_probs": [0.09061627089977264, 0.09797955304384232, 0.25413963198661804, 0.5572645664215088], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9c859f80463b88343717182bf5633a51b4c73363d7cfaf4d9b1494f0e772b34e:action", "state_id": "cfd58ce5df5eb34f3380b1a4c0463a62f529fc41a7ffbfaceb9e42641d767fcf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.994140625, -0.91015625, 0.037109375, 0.83984375], "student_probs": [0.08967841416597366, 0.09753531217575073, 0.25150933861732483, 0.5612769722938538], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "90702a668f7ba62e8afef41452765eb003235df1257211d930d6b66f4ba50636:action", "state_id": "e794b6079f592a9bc007e365afaeab7d70c60bf51f92f7df8f54702544ccc899", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.99609375, -0.923828125, 0.009765625, 0.83203125], "student_probs": [0.0906502828001976, 0.09744369238615036, 0.2478610873222351, 0.5640450119972229], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d4e5bd8ef81db48021667a906c24216d0e205897a8e5871e47dfd03f314b7f5c:action", "state_id": "ff7293348aebed878f63a1e4960282b84985c06e3b51a287abba94a23d851cf5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0, -0.9296875, 0.001953125, 0.81640625], "student_probs": [0.09135624766349792, 0.09801094979047775, 0.2488175332546234, 0.5618152022361755], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "48432c7bc95bbd61185978e229622e800159ce5d440598bdcc456089f8b7a615:action", "state_id": "729d1e4de206634dfc63365e0ba2f254631ff0d0677595251282e8b6773aeba5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.99609375, -0.943359375, 0.01171875, 0.791015625], "student_probs": [0.09288573265075684, 0.09791546314954758, 0.2544699013233185, 0.5547288656234741], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4a53365a4851041067da19dd80876ca1350a1123f9f6b32f7530925b0f9cf42a:action", "state_id": "e8153ed63c2c683bf24c0b1ec9ff3cad3c9ae1d6664cf9d50c51380cb4faac7e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.017578125, -0.935546875, -0.005859375, 0.798828125], "student_probs": [0.09102881699800491, 0.0988108441233635, 0.2503587603569031, 0.5598015785217285], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "34a0f49a44a84f0a536cb6e181e82b83342c4c75f5089f7d902140259c6c050b:action", "state_id": "a2681c747fde7620e937f228262532a1f43d99367d08f4f205b0027c582a2e1c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -0.9453125, -0.005859375, 0.775390625], "student_probs": [0.09247837215662003, 0.09921480715274811, 0.25384917855262756, 0.5544576048851013], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "11d39ab21fceb22a992f669415c13551b005809ec55d39164d56d7c236615830:action", "state_id": "092adc3384b10ecec8cffd0fb1d592c48f642f074708e8f877b16ad4036955c5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.947265625, 0.013671875, 0.791015625], "student_probs": [0.09043600410223007, 0.09778463840484619, 0.25562334060668945, 0.5561559796333313], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3d55e2fd31aff441590747c68f7f0a8fc55ed07b836d1efa793b744f8a17ec20:action", "state_id": "60a4c5a3a0f8cdb352c4fb7dbc0d69adebfb38de57b7666d468b51a1b6ea17ca", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -0.94140625, 0.01171875, 0.796875], "student_probs": [0.08933690935373306, 0.098117396235466, 0.2544971704483032, 0.5580485463142395], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "de15eb8cf167f95f14f0f4dcce4d0eba376aec9b831069f270aecb563487f5b9:action", "state_id": "c5a7cc38dea234977c96ced2b8242f6e58578a6fafefb60a6da132d02daae17f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.021484375, -0.939453125, 0.0390625, 0.7890625], "student_probs": [0.09019386023283005, 0.0979044958949089, 0.2604753375053406, 0.5514262914657593], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ee0ac6b02ebce55d0ec5432dd70e0dd6a19e38fecd905cee098fd319be64dc9b:action", "state_id": "01620d22c98c4f86443cea6f7a3d1a639ca52598506b90c6a69fb7821a5a0149", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03125, -0.95703125, 0.025390625, 0.78125], "student_probs": [0.09025698155164719, 0.0972105860710144, 0.2596414089202881, 0.5528910160064697], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6ef8f1dcb2ec16d439527372ca134df537bf1fa8a359a69efdc1d2401413bd56:action", "state_id": "b5a78962b4633a9d16827cf3c42025e61c05f3b95f5fdad87b64930aff0decf5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03125, -0.955078125, 0.029296875, 0.78125], "student_probs": [0.09014823287725449, 0.09728328883647919, 0.2603435814380646, 0.55222487449646], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1abd103fd4d470a97845f3c299601dda0b8253bc68e7d21bc06c89169321b545:action", "state_id": "300839f97fcb53a3df19cc87db50b8bf92dd6acfc8c38772f6b9e8f121da214d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -0.951171875, 0.021484375, 0.7890625], "student_probs": [0.09006669372320175, 0.09738530963659286, 0.25758033990859985, 0.5549676418304443], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "81d821d11d5adc4d8437c8b9aea229b28cc0c9ba63d6feb7a5fc0394152ddc32:action", "state_id": "ea92ce82d54837c4ae5eb89553ad7b63f9fba3238b6b536e7add8bfc94a673c6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -0.94921875, 0.0, 0.7890625], "student_probs": [0.08974422514438629, 0.09818047285079956, 0.253667950630188, 0.5584073066711426], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0b6eedeaa3858bc3489c2fa809a8cddebb0958ab8410af7ff4d4560a4cae3200:action", "state_id": "9ba3b260a24e3f88f10b45d9b3df80193708b4e51a42eecb7200a8e4b268fe9c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -0.951171875, -0.001953125, 0.7890625], "student_probs": [0.09060733765363693, 0.0979698896408081, 0.2531238794326782, 0.5582989454269409], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cd0967ea743b65731572f6e767c1a7c5969726f5e04f18b8ee206da612dc860b:action", "state_id": "c075e1cf398822731af5227f8c5905cdf400d498f17e953774bf53a3c095f9a2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.02734375, -0.951171875, 0.01953125, 0.7890625], "student_probs": [0.09027224779129028, 0.09741711616516113, 0.2571617066860199, 0.5551488995552063], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f986af71df8837c73b9f974cb6ca4936d6a8d0420b531c642456846cd0dd0c74:action", "state_id": "88ba12e91233d0d95974678ad43ecde6359c2359b1d8eefafe550daccb7531cf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.044921875, -0.931640625, 0.0078125, 0.7734375], "student_probs": [0.08970891684293747, 0.10046922415494919, 0.257058709859848, 0.5527631044387817], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "77d7badd68b9ee9df3d8f78f774c59021dc6f56312abedbf57453d24bcb6973d:action", "state_id": "6117b1f44383ec2c6ea17709f85a9ef748a78dca418d3bd910d21a20a5e5e0ae", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.056640625, -0.96484375, 0.021484375, 0.794921875], "student_probs": [0.08768029510974884, 0.09611007571220398, 0.25770673155784607, 0.5585028529167175], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ea76833a671612af4a946c5a381aa80e439dc11be9d5a032dbcf1268720d75ea:action", "state_id": "76286baeb622e930932771d9bc7fc47b1fdebb7981fd55161240aa3a0c20ec5d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -0.9453125, 0.03515625, 0.822265625], "student_probs": [0.08915835618972778, 0.09565295279026031, 0.254982590675354, 0.5602060556411743], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "36362946a66a0c09828175a20d66ecd5448199739a4b2d997ecac3f7c388edf3:action", "state_id": "0b225518abbb6c9cf56af191a028d274bff415583c75de5e8152e44b22230eb0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -0.984375, 0.03515625, 0.8046875], "student_probs": [0.08925404399633408, 0.09335491806268692, 0.25877007842063904, 0.5586209297180176], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "07c8d45f0ec57b3068e00b0606e692ed80d2b6a8274a672471dfcab5cf506f2c:action", "state_id": "520bd2cf858029e2dfa93cb769722053864314c4b3e57613a2a6da381dd034c0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.021484375, -0.96484375, 0.056640625, 0.845703125], "student_probs": [0.08720286935567856, 0.0922846570611, 0.25630348920822144, 0.564208984375], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0f72b5131cb2f2073d790d1a089ffe4b3c9fe4fbea3337cab89d47642bd94318:action", "state_id": "948272d33ebe24e82caf3163937d83966c2945b28ee123a735ea4ed46c911f2e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.974609375, 0.060546875, 0.845703125], "student_probs": [0.08626539260149002, 0.09147102385759354, 0.25754088163375854, 0.5647226572036743], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a87f07f36802cb86b723b3d797c7f70541ccedfa2eaf4d7fc4e1b599c4acf0d2:action", "state_id": "0e596d12d407600c222c18d28beabdca6100e7e8e65cd0c27b48c9752be5244b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -0.990234375, 0.046875, 0.828125], "student_probs": [0.08786990493535995, 0.09137024730443954, 0.25776007771492004, 0.5629997849464417], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a469b9a2ee4483ca5f85c6894a6068715fca29a5f55e65b676675093bf3ed262:action", "state_id": "4d89d4eec0c9a94465b1d5c2800f482699053757f3756638c31dabf733422497", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03125, -0.84765625, 0.0703125, 0.82421875], "student_probs": [0.08616911619901657, 0.10353457182645798, 0.25927111506462097, 0.5510252118110657], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "96cab600aaa79da16cce32a4e54f595db270151fdb2002d496a77e3a08a33ec0:action", "state_id": "4e52ffbbae2d5ce7fa595f6659288e25a51be00ad3d7e6523bb9a39d1c73af2d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.095703125, -1.3828125, -0.236328125, 0.611328125], "student_probs": [0.10389949381351471, 0.0779692605137825, 0.2453777939081192, 0.5727534294128418], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f34c48570bff50b48cfa757ce0e941d58a23e00be821cbe180bd19c3980c8986:action", "state_id": "7891d3721e2ea26c7a7f5f5907da2d2e85091515f06edd94cc9f60e870cf005d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.96875, -1.02734375, -0.04296875, 0.861328125], "student_probs": [0.09344656020402908, 0.08812849968671799, 0.23584410548210144, 0.5825808048248291], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "54eeb22ac571bbccd94df9a0ecd6fab41392307be562b0855ddb67a8c0d58a1c:action", "state_id": "fefbf7c241ac6f851f4b1cc9edfe848cdadb54ab8822360c44bae0c11ac502b5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.07421875, -1.08984375, -0.048828125, 0.7578125], "student_probs": [0.09075033664703369, 0.08934338390827179, 0.25302866101264954, 0.5668776035308838], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1c64b3bc0d8cfa71adbe0685012185a67caabf95eaa0007c314150a519296633:action", "state_id": "4ca7db420bb97e1ebd4f5b7238cf289769037b7e37f1d1d406e88695cc23846e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.060546875, -0.966796875, 0.03125, 0.8046875], "student_probs": [0.08668999373912811, 0.09521032869815826, 0.25830352306365967, 0.5597962141036987], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e9f12e79fd49ba14718fb25b362171d67b1d281a669fdaec12c77a42bfbf2baf:action", "state_id": "7f6e7ef33ccd29452b3502b463511e3746165ed67fac96bd0ef50d8e300df7ef", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.24609375, -1.55078125, -0.42626953125, 0.322265625], "student_probs": [0.11355605721473694, 0.08373098820447922, 0.2577836811542511, 0.5449292659759521], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "71c286681014fd03e3e43448f144a25a5980491e40f37797b7868a54aff82588:action", "state_id": "6e3f5c2f7284fe6b07c5b3e4abddf3b9d17611b9d6599cc3b4a865b3031ff773", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03125, -1.16796875, -0.181640625, 0.669921875], "student_probs": [0.10318519175052643, 0.089999720454216, 0.2413226068019867, 0.565492570400238], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "00ceb16256e0bb06dd77d1206b32f3d0f1c04b23fc905a871b1b0cdd0424b78b:action", "state_id": "b72ad0b9f601f7c6d875f6c37e9f683d2a5adccf374a15df1e5378cd1435a6f6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.12890625, -1.1640625, -0.171875, 0.6171875], "student_probs": [0.09707251936197281, 0.09371910244226456, 0.25277242064476013, 0.5564359426498413], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1452167ce5be2ef326f38cedb2350b1b059c9313ff7b1379717fa4bf60728566:action", "state_id": "108abd09f3ecc54fd267e4347b475b80ecbebc7b0cb026b6483c5dfee0d97237", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.072265625, -1.1015625, -0.06640625, 0.677734375], "student_probs": [0.09560241550207138, 0.09284219890832901, 0.2614014744758606, 0.550153911113739], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1a8007265927fb808190e3da21ec5f37dd5960bff9989cf7610cfc6ffa9b884b:action", "state_id": "80a4a9430a59aa962b8dfe52bdca79c46158cb353f06d902083940b884dff593", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.36328125, -1.59375, -0.61376953125, 0.482421875], "student_probs": [0.09763071686029434, 0.07753452658653259, 0.20658330619335175, 0.6182514429092407], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ce66ed5a8458e46f2bfca98f9b9957e3f09b5421f53fe4d52eb017af13a488ea:action", "state_id": "d062cb723d6168ae87e878addccc545ff42c7eb7605975e905f1aca6c9df5147", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.26953125, -1.28515625, -0.484375, 0.833984375], "student_probs": [0.080826535820961, 0.07957343012094498, 0.17723233997821808, 0.6623677611351013], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "51a515a63122406b872a5922682e7c8c0a5afadd8990cc0003a0778a78c90932:action", "state_id": "c6360dc3526bdbe0a108e4db16dfd798f5ff65ac0d0d1c2a961822d4ad8ee2bc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.34375, -1.375, -0.916015625, 0.685546875], "student_probs": [0.08999485522508621, 0.0872260108590126, 0.13803218305110931, 0.6847469806671143], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c1e629ce62d054a626887c008f0baecb328dd8392df862b18279cd4add0b0151:action", "state_id": "9550492839908ed8dca388d0295b2275592129b143d0438b79db41a861e4bd38", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2265625, -1.26171875, -0.67626953125, 0.75], "student_probs": [0.09159938991069794, 0.08843504637479782, 0.15881143510341644, 0.6611542105674744], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "647d56c18cd3201ea0f91f71994e85acf7e7e0bb64cf6f664d6f068b0df24988:action", "state_id": "edc74b325187ca4a7740c4b849f9d415e58fdc769c0cad883a3692a5f10082e9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.26171875, -1.1484375, -0.67578125, 0.66015625], "student_probs": [0.09301996231079102, 0.10417740046977997, 0.1671265959739685, 0.6356760263442993], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6a96a95f96d4a5bc7d1d6422004b9904050a631c6583b5c521562e3a4609f9a2:action", "state_id": "01483ab660fd600534d7781a969dc043d43d24a631543db95c27ebbca0a9307a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5859375, -1.4453125, -1.0859375, 0.3515625], "student_probs": [0.0931004211306572, 0.10715792328119278, 0.1534966379404068, 0.646245002746582], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "88ed46bc180f9a4f72273395d7bf781962769f9b115134b24f2c58956a8795b6:action", "state_id": "a20ad437ada0514b5186b8f8c94d0fd8456b6b49c5ea28c8483f3cbd5c3f0150", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.138671875, -1.041015625, -0.318359375, 0.66015625], "student_probs": [0.09600286930799484, 0.1058512032032013, 0.21804261207580566, 0.5801032781600952], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6d19e1460277c02a1988160b991553fe52ffea0fed9bfe9b2f632ee23ef4b4b7:action", "state_id": "610493077e803d591a5f80fce199fb5c1c26b03ee2bddf4e5f2db5f2ec3711c8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.009765625, -0.95703125, -0.12109375, 0.880859375], "student_probs": [0.09001260995864868, 0.09488675743341446, 0.21890145540237427, 0.5961991548538208], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a15302f9eef6aaa30ac6d218fe8dcd14ce77acccceaf3fe7303bf9f0415c2cdf:action", "state_id": "8d61c44873bbad68055f9ce72f9850402dc979ba32a892930a1723fc542fb156", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.013671875, -0.966796875, -0.150390625, 0.83984375], "student_probs": [0.09258285164833069, 0.09702599793672562, 0.21950723230838776, 0.5908839106559753], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "776de58982d2631c704afa95f287ea22168eb8cf184a6b8fbbb33a8403d58c6b:action", "state_id": "0421314c55feaf3eab1c9cb5ae672020dd457946edca8f8caa9a68f30b0ba9d5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -0.98046875, -0.1953125, 0.8046875], "student_probs": [0.09307871013879776, 0.09908177703619003, 0.21726152300834656, 0.590578019618988], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c42886f298ba17593c31f2c58f029ca1b159c3fc4389ce0339a0e593ab5ff05c:action", "state_id": "5dc8977b8c4c863ec0b491c615812b01bdd6f7ce663cc6884239d85f39d03dbc", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.044921875, -0.94140625, -0.2890625, 0.88671875], "student_probs": [0.08977203071117401, 0.09956285357475281, 0.19116422533988953, 0.6195008754730225], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d9abf9d5e87d064049d8a8aa68bd23fa7474bac84f7842540819c87d1d57fa68:action", "state_id": "2458da58bb925a1d8a00a0e84c948ce4bf0b0efc4b717b8919b5215299e784ef", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.048828125, -0.955078125, -0.33984375, 0.85546875], "student_probs": [0.09220927208662033, 0.10127207636833191, 0.18736247718334198, 0.6191561818122864], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7a7486faa50b4d5418636fe06ae8a7ae0e8c5ae2b0c077d06d6b6cc27550f30e:action", "state_id": "5f307ab9e9d2ff53925c7646a549b3f91bba9cee8e5b97dbca6217ad74151d8b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.068359375, -0.966796875, -0.373046875, 0.861328125], "student_probs": [0.09092073142528534, 0.10064007341861725, 0.18223562836647034, 0.6262035965919495], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ce13731a21837ee2b7e802b3146aa309065452c657edbe64b4b95e0251449a9f:action", "state_id": "2ced0cf404831ec475ee31045e4b912176be1dadc30c9cf751cceb77a1d05fc2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.087890625, -0.97265625, -0.4091796875, 0.853515625], "student_probs": [0.09039944410324097, 0.10144050419330597, 0.17820757627487183, 0.6299524307250977], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8633a632cc0d4c3ec26f0e6d4e2e53a91ad95cf4bcf543ded2eae7b8ea14aef6:action", "state_id": "88ea277262df4c002ecbda90b8d99ed9c0b7a64b248395861c1c0920ddfe19b3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.078125, -0.955078125, -0.404296875, 0.88671875], "student_probs": [0.08907520771026611, 0.10073848813772202, 0.17474175989627838, 0.6354445219039917], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "905b061b2616c97caf020a9202dcc80eda8b8e285b54e1fd4a44e4a5ec052e11:action", "state_id": "38301837db265056b6e27f48a6aa82acfca7704791beba8d40cf8cf69e27ce0b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.078125, -0.953125, -0.4013671875, 0.92578125], "student_probs": [0.08681600540876389, 0.09837542474269867, 0.17080947756767273, 0.6439990401268005], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f588227769828eea43f36e8cb7724568a8eeb44f330ebd509c4b2e107187ad6c:action", "state_id": "8e235ec28c09b9f81dc09c945bd0f455f063895ebef7a8cf7f970389748c763b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.11328125, -0.98046875, -0.4296875, 0.849609375], "student_probs": [0.08894503116607666, 0.10157841444015503, 0.17619870603084564, 0.6332778334617615], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "843c2f9269cdaef3f450a598ae1c192742fb64aae3717444c939d6150bb4aa99:action", "state_id": "bfe1a64e6ae918ddb9d598cf91ac927e5da3f333093190c620083b68e4da8a34", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.123046875, -1.001953125, -0.43017578125, 0.81640625], "student_probs": [0.09022725373506546, 0.10184227675199509, 0.18040470778942108, 0.6275257468223572], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9ce73fd393d285ac48a168930c8662614ea7a4b677fbdf7f0ea21d50bd9d30d4:action", "state_id": "cead1cb498aaf5a7721a7d0bdfe8b06d15b49ed7cb168fcf5b1dba560177c134", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.25, -1.099609375, -0.75390625, 0.78515625], "student_probs": [0.08727504312992096, 0.1014387458562851, 0.14333122968673706, 0.6679549813270569], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6473682c3c2eac98773efb76aff00aa21739e89e8d3f21029985a1fb842ff487:action", "state_id": "3387fe218707ee1df1442f91b31105e412b0a07d788776fcf841729fa7284c42", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.39453125, -1.23828125, -0.923828125, 0.640625], "student_probs": [0.08753743022680283, 0.10234162211418152, 0.14015789330005646, 0.6699631214141846], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7cb1dec85e6afbf3df73ca8f42f6d941def6da3f4920158469c283335408fa22:action", "state_id": "e45e93905bd87e89ab48d429f362515752f9c774fbfc407a1a88846cfee50211", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.54296875, -1.3828125, -1.1015625, 0.48046875], "student_probs": [0.08855145424604416, 0.10393233597278595, 0.13768798112869263, 0.6698282361030579], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4b021528c1ff2a8c8f646f63e3e3ffc6f7b09884a7f8b87ed53d5e8070b14f0f:action", "state_id": "44941e135876cd5eadd7c6a26680d94d82d3da00ae204525289fb31dab093b01", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4609375, -1.328125, -1.08203125, 0.91796875], "student_probs": [0.06946463882923126, 0.07933110743761063, 0.10146602988243103, 0.7497382164001465], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "97b4a6c77e327c99ca69749a140eec2978d379fff509c77f65c0e7d040fe475c:action", "state_id": "22219b2eb54be78d6774f24be873fd86dd17f7030554b81b28c98909bdb747ba", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.41015625, -1.35546875, -1.03125, 0.919921875], "student_probs": [0.07248666137456894, 0.07656117528676987, 0.1058802604675293, 0.7450718879699707], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ef8964c19c61e868c546c71a34eb7cfee9abf82e3e60d8d9fee02c1bd4a5fabe:action", "state_id": "d0279e7e85698126b16aef87cedde6807c8cd3938a039d001ab7a186616c3def", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.42578125, -1.38671875, -1.07421875, 0.822265625], "student_probs": [0.07733745127916336, 0.08041822910308838, 0.10991868376731873, 0.7323256134986877], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9f5ffe41182713936aca183d9316f2f098bce0e0664a1fc797385751ffcf66fc:action", "state_id": "c21be05ae8d65f5859f8326a991f3719c6cb9a40b423bcc264e9b1385198b1bf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4609375, -1.40625, -1.12890625, 0.69921875], "student_probs": [0.08249123394489288, 0.08712811022996902, 0.11497598886489868, 0.7154046297073364], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "556d6db969288f2472ea2c79eaa2aa0d31136cf02e884f85608dc0ee484ba056:action", "state_id": "924ff51f81da73888e2243e41728a66f003c49308227b878bde2718c7c25ca66", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55078125, -1.43359375, -1.2109375, 0.56640625], "student_probs": [0.08448231220245361, 0.09498601406812668, 0.11867467314004898, 0.7018569707870483], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "917bf77081fb2e7db3c092aaa160d062038365b619241bab1f7a17034b636d09:action", "state_id": "cfa37b8b9a3bfff7c96b473e913c0fb0cb62fc5062575e27740b7a45c986f010", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6171875, -1.4765625, -1.328125, 0.509765625], "student_probs": [0.0842074453830719, 0.09692216664552689, 0.11243169754743576, 0.7064387202262878], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7bdc3f247c78a7fc01f8b8a3f6e010ac76268236e5089be8e6749435ad7322df:action", "state_id": "20f186c7915470019aa71e762678eb1100a9bcd6e603a83d01dd4fa89c6d01b5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.103515625, -1.375, -0.234375, 0.611328125], "student_probs": [0.10306181758642197, 0.07855857908725739, 0.24578805267810822, 0.5725916028022766], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "93b43ad6cf0b4a143c6f6afaddd4ed9d519a6d1b3f5de124a02886a2c8660d9f:action", "state_id": "96968a62ee06cfdd4226f0557cf3c1ee0529a36876a60171d12476d19926ad3a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.96484375, -1.029296875, -0.048828125, 0.861328125], "student_probs": [0.09392351657152176, 0.08806081861257553, 0.23474420607089996, 0.5832714438438416], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "452fe5e8370a7f586b84b44ec786be2178069edc5389b6bb8ba35bff4bbeeb0e:action", "state_id": "47d1d55df3ebdd1016ac8983e2e4b5579b7541a3d6efe95a89938e9aac21678e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0703125, -1.0859375, -0.048828125, 0.767578125], "student_probs": [0.09053804725408554, 0.08913438022136688, 0.25145259499549866, 0.5688750147819519], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "50ec9511efc421ecd05dca45d850830c28cb84650d9a65c5ec13deb46a953b8b:action", "state_id": "9cb87a714ae1322083794b63b9d90445454f647c7ac9a69555104b5f80aef2e0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.060546875, -0.9609375, 0.03125, 0.8125], "student_probs": [0.08626298606395721, 0.09529810398817062, 0.2570312023162842, 0.561407744884491], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6355ae4b19040971240494266727f024cbc8d99d47fb310268f449d79a5d790c:action", "state_id": "6d820af95dd0a8cdc471c14598bf64d74f7c4475e399c863cb159581225cdd5c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.15234375, -1.47265625, -0.333984375, 0.416015625], "student_probs": [0.11374664306640625, 0.08257120847702026, 0.2578383684158325, 0.5458438396453857], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8570455aa10de5cb1cd742412eb3cde493df0df40d5d298c64891a45649dda34:action", "state_id": "ec808598a8e8fe9a24108315a92efa479fd0c2674722d9e803f79fb6b801d710", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.98828125, -1.107421875, -0.138671875, 0.73828125], "student_probs": [0.10154641419649124, 0.09014102071523666, 0.2374899685382843, 0.5708225965499878], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d1ba381397f286f4e1f90400d6e70b909eb08554bdb80df824f437c092c4e7bb:action", "state_id": "aa1a633a268b373c3cda9e0fb3297a30de60901cedc2540620b31af5797d12ae", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.080078125, -1.076171875, -0.119140625, 0.703125], "student_probs": [0.0946347713470459, 0.09500516951084137, 0.24738912284374237, 0.5629709959030151], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5e72b6313ec8ebe9451943cbeea9b1002ae9ef0306e47b84b7df15f41fb1a2ae:action", "state_id": "5f34b86849f346ed24e8bec1dc8adde1988dd8fc1365db1f7313d84d84e9844d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -0.962890625, -0.021484375, 0.78125], "student_probs": [0.09042706340551376, 0.09796611219644547, 0.2511443495750427, 0.560462474822998], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d2ed2266ecb337e96b4c9308aa2008545f4cba9c3423fb7f953b1983e22f0359:action", "state_id": "6378d67c2bf434574a2a575e8443e486d47287425b536ed5d021565f25d8eab3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01953125, -0.921875, 0.021484375, 0.82421875], "student_probs": [0.08885041624307632, 0.09796502441167831, 0.25163254141807556, 0.5615519285202026], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5a19e23aef5d752653600cbcb9a3120c225b3cf7dfb2284b6345254c94889b14:action", "state_id": "8c2bf8635178dada6d6017f3946ec4412eb447e245b3f2aa84b0e3ef5c73b734", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.017578125, -0.923828125, 0.03125, 0.86328125], "student_probs": [0.08686792105436325, 0.09540574997663498, 0.24794748425483704, 0.5697788000106812], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f4f8a1869020a9dff94d017a229d628b39df1d3ccd889cec150272c094e7480a:action", "state_id": "0387ccbcf4a4d4383d313e517bd4f131695001576616b3b2efa139098fa2510b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9921875, -0.919921875, 0.0546875, 0.857421875], "student_probs": [0.08864453434944153, 0.09528763592243195, 0.25252479314804077, 0.5635430812835693], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9752be725d6d583ebe3ef571e3c2b8273359048453a8f83cb43f4156be742fa8:action", "state_id": "159d135a117876c9966f3c7140b425d33c5a5e7e1fc77d174146e4d3b4916bdb", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.98828125, -0.90625, 0.048828125, 0.873046875], "student_probs": [0.0881926417350769, 0.0957322046160698, 0.24879589676856995, 0.5672792196273804], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "97d5787b26a66f939804a26fc5b1766dab66e58f12a47270f30a62455dbed609:action", "state_id": "7283918e6ecd621de1a9bd7ccf99c8f6212704c239421a56afe5cae953dfc131", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.00390625, -0.919921875, 0.029296875, 0.841796875], "student_probs": [0.08904549479484558, 0.09684693068265915, 0.25022250413894653, 0.5638850927352905], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6fc2bbb1e224256a982dd9f1a849b98d5f48aa32b31a3641e7ba917517c0c2cd:action", "state_id": "23d07661e93313d3449adfb7520ec3ea44d1cac547cf987b80e29f425d29a0fe", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.005859375, -0.931640625, 0.033203125, 0.82421875], "student_probs": [0.08978286385536194, 0.09669995307922363, 0.2537772059440613, 0.5597399473190308], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "99b3fd429034001b5c03fab18c25d916825f06c2961670ee9cb43da4d716d2b4:action", "state_id": "99d903ca6f9ac93f43a5818c04a26e15e69113d9d0518e1c715d3c3624cd4544", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01953125, -0.935546875, 0.01171875, 0.814453125], "student_probs": [0.08967841416597366, 0.09753531217575073, 0.25150933861732483, 0.5612769722938538], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d82772fa43dc80dc070faabcf4aaf8e9ccb6084717dbf2c805eb36ff1417535d:action", "state_id": "5997edcdc4e957d723ce8da967be5b6e08129ae67233adb8affaa128c4de4f32", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -0.9609375, -0.0078125, 0.78125], "student_probs": [0.09074083715677261, 0.09773173183202744, 0.25349682569503784, 0.5580306053161621], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "44d48ef302070f64974a5c09ca267e3ab1302e03368b526f0183812d904f1b7a:action", "state_id": "b69c08f672ae0c047e1cfbee18eef0ab33369998d33e715ac4ce6c5449d140ab", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0234375, -0.916015625, -0.13671875, 0.83984375], "student_probs": [0.09102986007928848, 0.1013529971241951, 0.2209433615207672, 0.5866737365722656], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5fcbfe7949a4154b218b7788c022e07375af076ebb749175125706671585a8ac:action", "state_id": "4abbabcf074775e5fee4168e0deaf62f8dc5e0327244af3fbf3f6beaa93e65ba", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.90234375, -0.212890625, 0.83203125], "student_probs": [0.09266407787799835, 0.10479727387428284, 0.20882171392440796, 0.5937169790267944], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3db23b6c1773ea94089443b7aea78762ad1a7d3f4d2da3dcd246faeef7555c76:action", "state_id": "2a87237fe3cf26f728586a944d40061ec0cc2af7811ec4c342d3095ca0a80a62", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.99609375, -0.908203125, -0.240234375, 0.841796875], "student_probs": [0.09519656747579575, 0.1039421483874321, 0.2027154415845871, 0.5981457829475403], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "61e34e2a4ece3504d9d010d0f6abad68c1fbcaa3d179c71715818b351d9aa50b:action", "state_id": "4943dbe11f23b6ffcb6b30271affb7e6f1217dc6a96d2677f4bb3afea9e479a9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -0.955078125, -0.2841796875, 0.84765625], "student_probs": [0.09281257539987564, 0.10055051743984222, 0.19667619466781616, 0.6099607348442078], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6bede97165afad47e35844ae2ddca928c3522aff76e12de531cf7c0c89879ccf:action", "state_id": "0efa7d33ec1ca89b137624236601df86b0f0d1b01783159d6b85176838a60a9b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0546875, -0.994140625, -0.25390625, 0.755859375], "student_probs": [0.09612161666154861, 0.1021212711930275, 0.2140897810459137, 0.5876673460006714], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e59d2520ed3e7ba48567df87b41113e2fb1b7b26ac152f1ab7b688bad12a138a:action", "state_id": "2911fbf4becc2593e78059ada7a032b6f2058e730ffa8888e828b2e130134359", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.07421875, -0.998046875, -0.208984375, 0.72265625], "student_probs": [0.09536994993686676, 0.10291830450296402, 0.22655731439590454, 0.5751544237136841], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4a7f45121c40af7aa5ad3d9138919e18fdab7233f01dbbe15eee330d4b8c69e1:action", "state_id": "2950f1941abc1e7288a51f942caf5864bfa5aa69c3587cfe905d2d491892a15f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.068359375, -1.0, -0.12890625, 0.712890625], "student_probs": [0.09463776648044586, 0.10133339464664459, 0.24213847517967224, 0.5618903040885925], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a5f595095bd90ae48efa4c2b8eb366088af627e9041a34da1325f47c5ed85de4:action", "state_id": "4db4defc0d2c47bc394fdb3f072aba1c5cf8e97a810fe15fa8643b2fa8cf2e3f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.96875, -0.00390625, 0.755859375], "student_probs": [0.09217003732919693, 0.0983063131570816, 0.2579928934574127, 0.5515307784080505], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0e201004c439621add05db2b2c3892dfb925f0e44f9befda48463b209ace3c09:action", "state_id": "5cf4e971587d2b9260b6972d2d1e58d5084231915fd5df98e6b8414aa35d11d6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.048828125, -0.96875, -0.02734375, 0.740234375], "student_probs": [0.09221242368221283, 0.09990033507347107, 0.25610288977622986, 0.5517843961715698], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2504bb5d9735f3dc161690e40801807647b7bd3a3792466cd2dd40cda95965ae:action", "state_id": "308de15307aa73fe5316aa4bc28341fb08ebe3dbcd570eac0c3887070f719c8b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.052734375, -0.94140625, -0.03125, 0.732421875], "student_probs": [0.09211845695972443, 0.10296647250652313, 0.2558419108390808, 0.5490731000900269], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "df115b279b9a31c36dd1d318a319b7dd4752b88cb18109f5a698d413e4d0adb6:action", "state_id": "1eb4d338badc2e23abb47b60a7b3cc81a344d586f07da7a808e9bf377e038579", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -0.97265625, 0.0078125, 0.763671875], "student_probs": [0.09072201699018478, 0.09733051806688309, 0.25945448875427246, 0.5524929761886597], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "850c8bcaabcf80724350a09198a50d987d96ddc4413f17888f4e47a24aa91821:action", "state_id": "40c8c6be1687f46bade40108e184f5fa1bea30a7474fb75e13845ff4263bf74c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.021484375, -0.96875, 0.00390625, 0.78125], "student_probs": [0.09167125076055527, 0.09663520753383636, 0.2555963397026062, 0.5560972094535828], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "600fd9d3c11183e4f1ba3fccdf5ecb4bf55c1ad7ae62db1120c3f7a363d2179b:action", "state_id": "2e314f74ff7f579035d332f8ed56e7f7a5ccdef2c7ec900cb5d063c742d0a193", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -0.998046875, -0.013671875, 0.763671875], "student_probs": [0.09144949167966843, 0.09565124660730362, 0.25597602128982544, 0.5569232702255249], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a302584a11a5e1b01ee2fedc438e8d07a2f091192d3b37d5b4b1a6ded23122b3:action", "state_id": "59df6237e543e4efc19dfc6a1b2691a075d3a503bac46d6085012a41d3ef58cd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -0.9765625, 0.02734375, 0.779296875], "student_probs": [0.09015783667564392, 0.09559835493564606, 0.26088035106658936, 0.5533634424209595], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "19a9f7be187777b37a0e884f49e443a5a19b66f0699e7094a4ae7be3731d6b09:action", "state_id": "aff92baec5bf86b80113a823dad53f8b5f3004192771ae2255dc7126bbb98943", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03515625, -0.98828125, 0.03515625, 0.771484375], "student_probs": [0.0904630571603775, 0.09480447322130203, 0.26381656527519226, 0.5509158968925476], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1d779bd5200c8b46ca8b07fa7038c49a70f6c9c05acca7e5a6caa230a2c3349e:action", "state_id": "9b5c2a21a84b33fe7f10c463bd0c3eddefa914fb3eab3e10c9e241fb2d6e52d3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -0.9765625, 0.017578125, 0.7890625], "student_probs": [0.08957848697900772, 0.0953558087348938, 0.25768962502479553, 0.5573760867118835], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5a83c6c1cf6d3c99f400df28093a6161c47bc9f8526ccc6622f75e6626936243:action", "state_id": "6b24f7c2adfaae5213e46f315e44bb39acd04c2a23db960c04cc0264466e0477", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.029296875, -0.97265625, 0.029296875, 0.8203125], "student_probs": [0.08851180970668793, 0.09366987645626068, 0.2551189363002777, 0.5626993179321289], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b4b30adb3005b8672a0fa01420552fd33737d96a2bdb28ba816c12b6923586ba:action", "state_id": "334707eee96d228f9d215dd2d9c05e6ed9c5b50e9424f563f5ed2f27e398be8a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03125, -0.986328125, 0.037109375, 0.810546875], "student_probs": [0.08877518028020859, 0.09285405278205872, 0.25838908553123474, 0.559981644153595], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1c60e1920ad766d765854797736d88526f59b33bf5a98d521b92021f2de6b1ac:action", "state_id": "2b45e592109b09d0ab6031e53d2423a799f3ae4533d99fe53bc2f16b32f90202", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -0.994140625, 0.015625, 0.818359375], "student_probs": [0.08814918249845505, 0.09237954765558243, 0.25357794761657715, 0.5658933520317078], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ddd8e44219766934e81065f9c513ecaf5a7ad14300597a54bcb61ba889760917:action", "state_id": "d5457162a7447980bbe73065df33a1b1a127a863dd82fa60ae982052f65e5944", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -1.0234375, 0.017578125, 0.78515625], "student_probs": [0.09016815572977066, 0.091588094830513, 0.2593859136104584, 0.558857798576355], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c1cf736e933fa111e5ea7f4ba361e2977a3b53f294ccff066d0c2b3acd09b490:action", "state_id": "163df2af7b97e6e1650328b6342a2d223320f07b995c26640ccc6d2963f6e3e5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.013671875, -0.953125, 0.01953125, 0.845703125], "student_probs": [0.08855675905942917, 0.0940842404961586, 0.24884912371635437, 0.5685098767280579], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1c184529db68b110cac1de4c1a2273842504d4c27354c387f3aca0a3d7c8e4d7:action", "state_id": "a10a42881b8340fed00c31a4eee1843b4e8b82ba4c936e506f4ad30b1328b019", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0, -0.931640625, 0.048828125, 0.853515625], "student_probs": [0.08843865990638733, 0.09469570219516754, 0.25243082642555237, 0.5644347667694092], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b46848ab63aef7f592721e781bf4db07ce86851b160159a08426cd3c0d5ff7b5:action", "state_id": "d60ddcf85eae164954c75d29871cd72040f69233b3cc74c686ceb6c808c54a5e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01171875, -0.947265625, 0.033203125, 0.822265625], "student_probs": [0.08953732997179031, 0.09549833089113235, 0.2545704245567322, 0.5603939294815063], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7b4067f97874ae86f8d91e06a2027d4e9952c4fac12fddbd3d572eb88a9dd045:action", "state_id": "52fb87b1dfd825e6662dd211387476d92673c807b2ca67b13a680da1390bf521", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.001953125, -0.93359375, 0.044921875, 0.83984375], "student_probs": [0.08906823396682739, 0.09536980837583542, 0.25373178720474243, 0.5618301630020142], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e2b124dace8e6c7c4ffbff31e45ebacebaf992193e4a408318590707ac6b2542:action", "state_id": "3ba3751f1b69964f607eec2aacd85a0b43333fc054af5c203566b3fa717b02c7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01171875, -0.95703125, 0.037109375, 0.822265625], "student_probs": [0.08953121304512024, 0.0945638120174408, 0.2555493116378784, 0.5603556632995605], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1a092230dc09d5493a13f62e5c084488f1d24a647fb5f6a8ae3f93fa2be7f32c:action", "state_id": "727666f1fd53a09f2ba1c9092501bfe0c1884284c4741c5e8d5fec6cd68dc4f2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.017578125, -0.80078125, 0.056640625, 0.833984375], "student_probs": [0.08665841072797775, 0.10763770341873169, 0.25371024012565613, 0.5519936680793762], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "554e71d9869b9a1aca92cac4fa47675d4e2048e5cb73f34e9c1371c936e2f6df:action", "state_id": "7a765b8e390920e5ed7a92c8f54aa1d4b78f8f38ff06a5ef7b5e44250c520770", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.26171875, -1.5546875, -0.416015625, 0.296875], "student_probs": [0.1132785752415657, 0.08451096713542938, 0.2638954818248749, 0.538314938545227], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f7b856284d2075e2a7780ddc98f3a17be5bd6a2f78b055542f4fa595c5a7fa2b:action", "state_id": "bf59a20a10471999e054b530f83aec3dd7fb1559e5d0dc621a2d6566487bafea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -1.18359375, -0.203125, 0.638671875], "student_probs": [0.10479450970888138, 0.09086939692497253, 0.24223105609416962, 0.562105119228363], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f3fc874a4f6de3c1d8b81feef63b94a5a7fde7314c7b07161284e611f1438db3:action", "state_id": "c120ad35840646c8d2795d97926c88b5dd94adb7e58eb291282d41221a4df356", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.119140625, -1.140625, -0.203125, 0.634765625], "student_probs": [0.0975116640329361, 0.09543903917074203, 0.24371211230754852, 0.5633371472358704], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "06fba4c1d6bdd78d467a1cc23c7c73708c28d0b18d684e7171578e83e7463283:action", "state_id": "8ae23832529ee523ecadc7d1e560a9a9b107f5ad27e71b6bf16deb38471b6fb4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.087890625, -1.01171875, -0.109375, 0.6796875], "student_probs": [0.09437253326177597, 0.10184192657470703, 0.2510785162448883, 0.5527070760726929], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "53b9afe9a82a15471c9ffefefef2494024d3ef321c522c20acbf9bd8909c71a4:action", "state_id": "730be4186f2180e9d57f02d811046b7e3f410478ab90d63946a05ffc8a3860b9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -0.98828125, -0.0625, 0.7265625], "student_probs": [0.09459568560123444, 0.09971800446510315, 0.2516722083091736, 0.5540140271186829], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b78a923b61e43c0d1bf8e984c67f3df192dffd472d34421202a0c71bdc2d2a4e:action", "state_id": "51ed66458add52952df90d746a30374e2b5dbaf91fa1da8dea7d69c7a88eb0fe", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.515625, -1.67578125, -1.375, 0.74609375], "student_probs": [0.0793488472700119, 0.06760605424642563, 0.09132995456457138, 0.7617151141166687], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ce0c8935cfd98e9dcf28df1c895b45e25216a734fe1e3a85c5b0b6b6a739c667:action", "state_id": "a1da3d52c8587cc0ebe41bb5f484636239fa61fb699da1e3c5e6cfd8cafd555a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4453125, -1.41015625, -1.16015625, 0.55078125], "student_probs": [0.09323139488697052, 0.09656736254692078, 0.12399493902921677, 0.6862062811851501], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "83d8fb07ec071d53ec32146698f76256567982b6b9bb45933f68d621bfa246ee:action", "state_id": "1eea23b153f258b985a337de2a3c0b6ddfe4acc10c66a52630b9c89c76cf6ffd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3046875, -1.19140625, -0.7890625, 1.083984375], "student_probs": [0.06805665045976639, 0.0762198343873024, 0.11397344619035721, 0.7417500615119934], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f1374d6e7328cbee2a6c6463dd3264898c0eb8da99c1dcdcb241b797b66046ca:action", "state_id": "f6c25dc716b7ac05a5ce43562d0df82b53bbccf17000a79739d9312d88fc7593", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.16015625, -1.0078125, -0.7158203125, 1.078125], "student_probs": [0.07632879912853241, 0.08888950198888779, 0.11903127282857895, 0.7157504558563232], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "905812fd1451a01ba2f36b65948cb00a5601b5118b82a4f4a39b6f42dbeae371:action", "state_id": "b5db2ab737ecc226363e028e3800b3dfb591c00bd1ad821c72bb4dce207410e8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.15234375, -1.115234375, -0.563232421875, 1.14453125], "student_probs": [0.07255180925130844, 0.07529474049806595, 0.130766361951828, 0.7213870882987976], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "488e58131d313f2196302006c6e031b12baeb2eb4f95ca19bdf28a8d809808b3:action", "state_id": "528cb93212e3b51c5970f640508bdabd73a2e12b7ad9bd6d8144c9d955748759", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.990234375, -1.07421875, -0.05859375, 1.056640625], "student_probs": [0.08195525407791138, 0.07535339891910553, 0.20805738866329193, 0.6346339583396912], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3292176ec630714b5493adc08f574b4b68132edbe53ec44851b7ef9b177054bc:action", "state_id": "93847a77631450cbf8c502ebb2db6434833bab089d61808bc5f097d6cd469327", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.35546875, -1.6015625, -0.4913330078125, 0.419921875], "student_probs": [0.09942938387393951, 0.07773875445127487, 0.23594138026237488, 0.5868905186653137], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "48c62341a54fbe7c21b812e73ea7598c04574d8133414526568c90769cc28337:action", "state_id": "dac90d74719e3ae0aa544c45e175956a2a51b384f90483bd9cab8af15cdf44c3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.16796875, -1.40234375, -0.30859375, 0.49609375], "student_probs": [0.10600553452968597, 0.08385728299617767, 0.25035160779953003, 0.5597856044769287], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3ce70a5a52da8c014d30507002309dfe1949eae68e11c72254108a71d95d92d4:action", "state_id": "5ccacdfc87ad7121f70dc6403eb9e5ebb309e1ab4c17a7adf646252e32a1e9cf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.19140625, -1.28125, -0.3662109375, 0.46875], "student_probs": [0.10574700683355331, 0.09666058421134949, 0.24134916067123413, 0.5562432408332825], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c2139525568b5f4db3d4fadfe94f8ffee84ef2156361507db49414f4376b8b2a:action", "state_id": "57415a260176d1db7dd776b1f08f858f2f92c514d8a1d17109fe84678b144489", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.103515625, -0.970703125, -0.154296875, 0.6640625], "student_probs": [0.0944967120885849, 0.1079186350107193, 0.24415025115013123, 0.5534343719482422], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "be37abf02f80b375aa8e97e72da8ae7b30c36ff54b475fd069691b776840386c:action", "state_id": "3b353c964a07e48f2d4d37f94b69153db350edd8f13f81375d497dadc5aa4412", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1484375, -1.46875, -0.3583984375, 0.4375], "student_probs": [0.11346524208784103, 0.0823669284582138, 0.2500186562538147, 0.5541492104530334], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "626b71989a61682ce51f8e29cb1247c72d0235869172948ead2e1ca5a24375f7:action", "state_id": "a17deee6071bf6b49b265568a9abf0c0695e777684e14a9d334dcbd3d4739fce", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.978515625, -1.125, -0.126953125, 0.73046875], "student_probs": [0.10277077555656433, 0.0887671485543251, 0.24082329869270325, 0.5676388144493103], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ee72709acbae9af6a07dbc7eea25ee173c54b142a435006eda55955869f36a88:action", "state_id": "a15f930492121084e68b4613a1a5719d8a31bd12ce6e68d8d97240558a2d3927", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.076171875, -1.126953125, -0.140625, 0.677734375], "student_probs": [0.09731205552816391, 0.09249380230903625, 0.24801017343997955, 0.5621839761734009], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9077869268ddc2560bf685a8bd95da1596967e7acb2b82d7c90e9b2dd9a3e02c:action", "state_id": "1af447b9373d3b7b45f5188229e9a53b0c87710346b64fd4d4c703c5c44b96ea", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -1.015625, -0.03125, 0.755859375], "student_probs": [0.09473542124032974, 0.09473542124032974, 0.25352513790130615, 0.5570039749145508], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "31ed72b686bce13ec00969be78a79a4a20fc9c5a90bc5eb48f4943622269969c:action", "state_id": "12687c7ee4c50963520e64881af9c273151cd6c70b9040f3f33630ebaac35b33", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.2734375, -1.296875, -0.673828125, 0.70703125], "student_probs": [0.09054376929998398, 0.08844633400440216, 0.16491708159446716, 0.6560927629470825], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "243d32ca643b3f0c54e6dc95a8972129a7819e893ca56c02e220a367f0fdb9c3:action", "state_id": "0b95b23d78702562734f5b7f7c5044ee62b1e7a517b4df809f00c0efefaccf6f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.10546875, -1.111328125, -0.6708984375, 0.9921875], "student_probs": [0.08557607233524323, 0.08507611602544785, 0.13215507566928864, 0.6971927881240845], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "75ecc743636a3248bbfbde761b88bec68afc64ddb7e1801bae9f82bd5fb3625f:action", "state_id": "b5a1748321f6eca73a2308b3ea9234904d8b0856f31609ea94abd296daa5b76c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.23046875, -1.20703125, -0.7666015625, 0.89453125], "student_probs": [0.08342494070529938, 0.08540330082178116, 0.1326633095741272, 0.6985084414482117], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5821d5efa2404853ff6212b4d47e7c15783e0288a5a3db54545f398fb3f70bed:action", "state_id": "f04885292691b99b1668c7f5015260f9a1976a2041d0bc24c6fd1cc8c84699e6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.625, -1.40234375, -1.27734375, 0.427734375], "student_probs": [0.08730340003967285, 0.10907609015703201, 0.12359941005706787, 0.6800211071968079], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4065280e388c1ebee3d29b9137b117b4f6ebfaaf26f4a73a8580573b891ca8a1:action", "state_id": "5811b3cca5e8b04e595a54326c811c805884ef46b51d2be4d8bb3ee376ebb770", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.103515625, -1.3828125, -0.232421875, 0.619140625], "student_probs": [0.1026144027709961, 0.0776088535785675, 0.24519948661327362, 0.574577271938324], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "13afd51e3aed3a9fea1579ccf4a335d0ba28ffdfb518d59454fdc5a5e56dfd12:action", "state_id": "9240c447f858e9b509ec2efefabe3088b496c953156f5d8dd9765acf96e8145a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.96484375, -1.029296875, -0.04296875, 0.861328125], "student_probs": [0.09379413723945618, 0.08793950825929642, 0.2357984334230423, 0.5824679732322693], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c5d18638a49cc9aad8b6ae560d5673c7a99dadd67086d87fb2262e4c5590df46:action", "state_id": "082dfd836ebbd795ae024b389382ad3e4ca291cc10cd505b4e59a965772b863e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.06640625, -1.08984375, -0.04296875, 0.7578125], "student_probs": [0.09126144647598267, 0.08914737403392792, 0.25395724177360535, 0.5656339526176453], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d484aa163d263acab0b6be4cd0a843277033216c6df2c60e9d776233eb0a9601:action", "state_id": "0dbb423342a59ec2e5261c4453e5cb7280979d910825497ae88c731f0ee28538", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.044921875, -0.943359375, 0.056640625, 0.837890625], "student_probs": [0.08555985242128372, 0.09470612555742264, 0.25743794441223145, 0.5622961521148682], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5c5b59dc50dffcb3e7d15715f0adc963a18d981a457bd9a094ed9b64159e788b:action", "state_id": "751c80800b45193847201f243047a5af4144e6ffef88450c393759a7193131b0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.484375, -1.64453125, -1.3828125, 0.701171875], "student_probs": [0.08435671031475067, 0.07187281548976898, 0.09337437152862549, 0.7503961324691772], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ca963e869de744f0d26607d5b23e7dc179feb7732bde4f0616865b56a2535474:action", "state_id": "f7a73017ac00cf86e357166694879b5807ee1199704a1bedacc5d2befb6f328b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37890625, -1.36328125, -1.14453125, 0.609375], "student_probs": [0.09449261426925659, 0.09598065912723541, 0.11944986134767532, 0.6900768280029297], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "eb2c8307d96541787d1c7de1b54c531fd49af6c6cde620ff69169e1f248cf8b0:action", "state_id": "14c087fea1ca82242133ab69c831efdfb95ae00482ebab393e62c2eaf94d2e99", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.50390625, -1.4296875, -1.19140625, 0.41796875], "student_probs": [0.09729860723018646, 0.10479471832513809, 0.13299141824245453, 0.6649152636528015], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4d3756c76eca0e73700ed171bdee394f08d859d97764a2ec1e50fc813618a12f:action", "state_id": "bf18f7cea41735371b47fc98db8e59e5f842f62c71f9b7ad500233b4864bae23", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.4375, -1.30859375, -1.11328125, 0.4921875], "student_probs": [0.0960809513926506, 0.10930010676383972, 0.13287512958049774, 0.6617438793182373], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "262bb7778362b11222c104790e5d2064d9eb604bc6713e459d323e6134774087:action", "state_id": "181e19c1209e201a998078fc15853668a9b1d34444368900d5e3a81af5f9add3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.10546875, -1.40625, -0.197265625, 0.578125], "student_probs": [0.10411270707845688, 0.07706835865974426, 0.2581852078437805, 0.5606337189674377], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8f03f3a6344f2b60d12d88f88c420d74b4e493e73f0daf547ea03627fe0571cc:action", "state_id": "b8ce28b929a3053f39d92959f6047dde5a57a85723fd795716a7c8b19abf0359", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.974609375, -1.03515625, -0.033203125, 0.845703125], "student_probs": [0.09364505112171173, 0.08814336359500885, 0.24006693065166473, 0.5781446099281311], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d20ae4558565890f19f9d9a2569a1a39b23db7df675d2f238c1b4a826e1144f1:action", "state_id": "0de7171d29ddd1648c59a231250f4ac83c63587cdf69cd4778294303dd10ebc5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.056640625, -1.04296875, -0.03515625, 0.78515625], "student_probs": [0.09009867161512375, 0.09133895486593246, 0.2502323389053345, 0.5683300495147705], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c4686d9f93de7f4fa1c6c99a0131c208dd9db8f51ed6c67c7502aae6c61989f8:action", "state_id": "bc9e8c0d5b00c8d104c6c475e4795d3c16d1da086dbc486a7bf6f116cd42a56a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.037109375, -0.955078125, 0.017578125, 0.8203125], "student_probs": [0.08799900859594345, 0.09552202373743057, 0.25265198945999146, 0.5638269782066345], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a0893472e1f25b7753344a8e1d56f26f0ebd75c0bc7608f2afc0f3ee588ad9ab:action", "state_id": "2f167b2a6984a802e881e5da054c9f1e160b497c401f1e07a45f341d1319139d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.021484375, -0.923828125, 0.029296875, 0.83984375], "student_probs": [0.08775977790355682, 0.09676250070333481, 0.2509828209877014, 0.5644949078559875], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "17b2edc4cd662041284790cd4439b6a8c55f1dd6da6f77e45a1e64b917124a30:action", "state_id": "8ff04fd518d1b251465813e4a6284517a39d69abd232e58f93abd4a923ce2891", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -0.923828125, 0.037109375, 0.82421875], "student_probs": [0.08883224427700043, 0.09737277030944824, 0.2545466423034668, 0.5592482686042786], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ff021d5fe68176709b6325abcf55d48f79ba33f8e432f6b8dd3bbbb8f8a8e7ac:action", "state_id": "67ff9aba49a749311688f1a3de595c1b3d48833e631d0981427a13d9370c525e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.99609375, -0.921875, 0.033203125, 0.83203125], "student_probs": [0.09010349214076996, 0.09704528003931046, 0.252208411693573, 0.5606427788734436], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f49a7a1e9cc16c21d1607e431bfe5cb94e301f4ae1d6189d5f8b513058191969:action", "state_id": "64875a84c0d0043883651a3782c5b108a04abffb02f84a74b0abecddaf7f3286", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.994140625, -0.912109375, 0.01953125, 0.841796875], "student_probs": [0.08999117463827133, 0.09768449515104294, 0.2479887753725052, 0.5643355846405029], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "95877e27f86614b13933f9e5a318fa57e0bc78d1b1da08853f2175a27b42ffc9:action", "state_id": "a6468e619e2b870ecc2b345fdcd5a882e7151bc174f7431fdb312b01029f9816", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.99609375, -0.92578125, 0.001953125, 0.83203125], "student_probs": [0.09084276854991913, 0.0974600613117218, 0.2464544177055359, 0.5652427077293396], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e1e010aa6036f11e76e4c6cb83b4caf75193b3494b7d980b62918e10c615d2ac:action", "state_id": "76a9a7ba1730eba24de3fd128b0f3e42b63398e1c48471a88aa85f3ebc2a6d64", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -0.91015625, -0.12890625, 0.865234375], "student_probs": [0.09011019021272659, 0.10013327747583389, 0.21871118247509003, 0.5910453200340271], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a5f0fe08b1efe35b8773b2687e1ef3c5b43f988e0f178f43ad9dbc0aafd0e123:action", "state_id": "654af4d4041e72dbab6e0967ea276b0b1f2ffdde7096e4c5c4ffaab5e4f21273", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.017578125, -0.892578125, -0.228515625, 0.873046875], "student_probs": [0.09125763177871704, 0.10340844839811325, 0.20088832080364227, 0.6044455766677856], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ce9566bbbdd7a19ef08eddf5d0576f554d49925ed14b50321d055d3843650135:action", "state_id": "950c7fa3f944cab9d51d3e5326afbe66ccdcf526905229ea6a4002dba69662ec", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01171875, -0.919921875, -0.28125, 0.833984375], "student_probs": [0.09519506990909576, 0.10434732586145401, 0.19763004779815674, 0.6028276085853577], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "79f3cc81f2345d175bd858b63f0a18091e33aff665292353619a1875ea88b82d:action", "state_id": "a5c5d8d5caf4af238648396f7079bfcfc3d6954e6c96b2dc4471b269a7181622", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.05078125, -0.966796875, -0.3603515625, 0.77734375], "student_probs": [0.09704598784446716, 0.10554836690425873, 0.193565234541893, 0.6038404107093811], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1d2f4cfcc483876b2ea180461fbe7d9be9029b5f9f2fcaa82516eaa5758b23a6:action", "state_id": "3de7cd97a95fefe44822e81231032a01bcb648796abd13124ce602ab38b73af5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.951171875, -0.96484375, -0.1953125, 0.873046875], "student_probs": [0.09695734828710556, 0.09564078599214554, 0.20646491646766663, 0.6009368896484375], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3045932eb5b78f1c2a845594545fbe33957e10733bd7f28005b03ee961ffeee5:action", "state_id": "0e07891eec823c185dd5441d4f5c9faa5a83abf5a91a1ec485c21afe5729b94d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.974609375, -1.1015625, -0.3466796875, 0.8984375], "student_probs": [0.09744121134281158, 0.08582377433776855, 0.18257826566696167, 0.6341567039489746], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d772099d4ceca241c018e6b1cf8d36e30f912622a57398047f29dbe332ee5613:action", "state_id": "34d31dd6ec2301d59a7c850a4279f607307ba80fb424870fed1769fcb8885ab6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.052734375, -0.34765625, -0.68896484375, 0.982421875], "student_probs": [0.08253415673971176, 0.16704952716827393, 0.11874539405107498, 0.6316708326339722], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9d4ee44f06edefb5d22d334dd60a2e52fa8b323b965df57239a5044d310ddd95:action", "state_id": "d376e6dea3d46e9088439039de498d9c802e5347b0367a16ddeff890b2ef4e8f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -0.302734375, -0.6767578125, 0.966796875], "student_probs": [0.08333155512809753, 0.17469853162765503, 0.1201857328414917, 0.6217841506004333], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "89f84f3107f63bb15fdc0c885c7dc3285b548d2b40d3e72879077d9c3bb65639:action", "state_id": "fdb13470dce72c04ca548735e7686efd6fb00eb648400dbed6ccb369f1b92355", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03125, -0.267578125, -0.68310546875, 0.966796875], "student_probs": [0.08377161622047424, 0.17978578805923462, 0.11865721642971039, 0.6177853941917419], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ab8e677f95357f285939bce660fd0d1f51884ee0234e06925621e0ba4f5f7cb9:action", "state_id": "2ab71724db23a20c337bd8a5f9fca3a8bf0f7ec9825264a5a08476a23e23fb86", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.03125, -0.267578125, -0.68212890625, 0.9765625], "student_probs": [0.08325720578432083, 0.17868179082870483, 0.11804381012916565, 0.6200171709060669], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e3159858ddb88ec36a23560d27c7e5bf8698fcb9327f39ea3a17fbba2517eea3:action", "state_id": "cbfb00a6e300f467692c6d88f025b563c0d15c047f25e74a8e67d9a7a307ef70", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.060546875, -0.2783203125, -0.6845703125, 0.96484375], "student_probs": [0.08182087540626526, 0.17888784408569336, 0.11916498839855194, 0.6201262474060059], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "966cd8dc872bd150776a8f69ab3ff914ca4481160f1ca07e672646366edb8056:action", "state_id": "a481cf1019f9c0cb96aa00bb9318a51030fa64e5def2c8800d1ef818da12aca9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.048828125, -0.263671875, -0.6845703125, 0.958984375], "student_probs": [0.0827869102358818, 0.18153095245361328, 0.11916722357273102, 0.6165148615837097], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b13f669ab4b7d33ce820bdd96417aa5fd6f07eff20d80c28a5c600a733f524c7:action", "state_id": "30980d7c33b64d86293d62effcae9bc4f860e5600c3399bef56d3e5e06061c8e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.220703125, -0.6669921875, 0.958984375], "student_probs": [0.08374134451150894, 0.1872454434633255, 0.11983684450387955, 0.6091763377189636], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7831da26a544a7b8fb79bae28a72e041eace52c61b157148f99aa1d5b69e32e0:action", "state_id": "4247cf4c9ccb53fc90d66e0257b98776fc4c53a7f52c21943027ce368faee566", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.06640625, -0.296875, -0.720703125, 0.939453125], "student_probs": [0.08330406993627548, 0.17983300983905792, 0.11770724505186081, 0.6191557049751282], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "23212cb2b551c1516c74c8e3d52e79ab94128c827c249e6276c6315bc5a6cf8d:action", "state_id": "42ac712d22254a1cdc13e88dc9794fc1953da4619780b79db34fc9bef571caa2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.06640625, -0.25, -0.71484375, 0.9375], "student_probs": [0.08263358473777771, 0.18694649636745453, 0.11744600534439087, 0.6129739284515381], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7eca966f46e85418c4e0fcf037c3cc461c78841d2f0ca03f0e43662e4859bb1d:action", "state_id": "99bb69a6a9143a3b43524b22b258766149e5ac2726ac6f12ac62181167458d77", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.10546875, -0.3232421875, -0.7216796875, 0.91796875], "student_probs": [0.08184341341257095, 0.1789371222257614, 0.1201326996088028, 0.619086742401123], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9bace8e0d77c1ca8d1c35c56ed3c0ff875d5d845b98a0d0bcd6fc6054c3dde5d:action", "state_id": "0e638a14991e8646d36a86acfc0d4373adca6c7dfaea77dde4d7085d0b102767", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0703125, -0.3212890625, -0.67724609375, 0.947265625], "student_probs": [0.08253180980682373, 0.17454931139945984, 0.12227226048707962, 0.6206466555595398], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f0ecf868051570772b71851a6751c1699fb4cfec2743cbc6aab3e3376e15f2c0:action", "state_id": "66110404bdcdb050829aa9c74aea5bc92c6652d9c184f5c890967c0a79b2fe81", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.08203125, -0.3447265625, -0.68603515625, 0.9296875], "student_probs": [0.08297161757946014, 0.17343507707118988, 0.12328450381755829, 0.6203087568283081], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "37ea7ddf02dc20c4d165e149ae1fa455d69a64fdf1ddf24207c0654f5cd5b193:action", "state_id": "c239c87353ec41b793cae1d61c686fecc574fb1165a8b56266ede7e932068aec", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.08984375, -0.365234375, -0.7021484375, 0.927734375], "student_probs": [0.08293526619672775, 0.1711721420288086, 0.12221181392669678, 0.6236807107925415], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0b2604c2ea1078e320dee3d295d25a9735297ec01e5b34f297b5ddd6773f6615:action", "state_id": "d7f68cd8f101e6857178c3a58087e44348013a1c8c150b1a7fdfc27dde51e132", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3984375, -1.53515625, -0.7353515625, 0.498046875], "student_probs": [0.09546158462762833, 0.08326306939125061, 0.18526919186115265, 0.6360061764717102], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "33b00fc55261c3834fbb1417679db30f83d45e30bd18155d558fd9c641130aa9:action", "state_id": "d944f0e5452c5ce3a5bcdcf147769ca77d36364d226059a3c54eff042e712c7c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.21484375, -1.19921875, -0.669921875, 0.66015625], "student_probs": [0.09745600819587708, 0.09899071604013443, 0.1680603176355362, 0.6354929804801941], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e5bad8ee881330d6475533c18483ed141643c0e7c9c5fc687af353609a91ef15:action", "state_id": "c40394417e74300a9e8a0baff3af86ee87c9a44a9408869ef04858b47c042daa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55859375, -1.5, -0.798828125, 0.208984375], "student_probs": [0.09945526719093323, 0.10545683652162552, 0.21261298656463623, 0.5824748873710632], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "996f7f50c9e168d0f398a7e64fc458845f1d4d08e9e8112de747d4b9864aca6a:action", "state_id": "db1e075dcf09cc41d61f363dc04922287c4d067e3703d094a7c34770aaefabe0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.083984375, -1.052734375, -0.158203125, 0.685546875], "student_probs": [0.09593535959720612, 0.09898067265748978, 0.24212543666362762, 0.5629584789276123], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d176a8b8fef6651363a1a40ba59f537c07ee6e8a2eed3554469a88fc00437e77:action", "state_id": "94a2a406da2f1a337d3d00757ff02a287d34e922ac251a8cba0deb6fa4d3ef4d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01953125, -0.970703125, -0.0078125, 0.779296875], "student_probs": [0.09222602844238281, 0.096841000020504, 0.2536514699459076, 0.557281494140625], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1de0f5cc13fde3e913555a6e8eec531db2d84be604fff13644cfd3917722a6d1:action", "state_id": "a981aefe3b5021070a41d0122b166bfe35bf6cc6d2c3eb5bd09eeecb3a0cc22a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.013671875, -0.943359375, 0.025390625, 0.814453125], "student_probs": [0.089914970099926, 0.09646467864513397, 0.2541505992412567, 0.5594697594642639], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3545928684a49de5cc3a86b45bb8f296f2f1be03b607b5db2645e87b64e465de:action", "state_id": "a6cbba528ccb89574ea5174d5f70c123961a873851b9107094db14c5ba430751", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.021484375, -0.953125, 0.025390625, 0.830078125], "student_probs": [0.08857984840869904, 0.09484687447547913, 0.25234049558639526, 0.5642327666282654], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "38a359a0785d9675900a01c1f78c17f0ae8fecbdecf4b1289f0d155bbed6adbc:action", "state_id": "07f6e0117bf3192952792c78b5f627c37517bab471d8ed8316e66b193fc73de9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.009765625, -0.951171875, 0.029296875, 0.830078125], "student_probs": [0.0894257202744484, 0.09482206404209137, 0.2527677118778229, 0.5629845261573792], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6bc671c56054e5f9f200a046ae20bdc6afc335c97ddc6d4383df4b4a80ea45d2:action", "state_id": "65976808978c86676a483cba3e4f01aeb9d9eb771ee2ee34df6eaf8eaae3b310", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.015625, -0.9296875, 0.03515625, 0.814453125], "student_probs": [0.08941347897052765, 0.097437284886837, 0.25571224093437195, 0.5574370622634888], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f7da38eb0920526c330aa16a4412a990951e256b2c6aff5e2f4bb5d9d5900e23:action", "state_id": "1978f56185a023bc5c687809d1848d368986f26bd7af9bb64285ea0f4b6de6b4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0, -0.912109375, 0.033203125, 0.83984375], "student_probs": [0.08930584043264389, 0.09751025587320328, 0.2509540915489197, 0.5622298121452332], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7044afb8738f683e28862e372772e4f7d0d036172004214dd1089c6d9a280c82:action", "state_id": "325cba83611cf0822149bea1c54f9c2a40149dc45c261ed6cfb1515bcf6c1d2e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0078125, -0.912109375, 0.046875, 0.84765625], "student_probs": [0.08798011392354965, 0.09681615978479385, 0.25259774923324585, 0.5626059770584106], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "98558ad4c6ceed8ee7606d5fab77181c7959b740efc8356f73ade6cb23aff942:action", "state_id": "ed629d503060d02b213fc1acc7d33bb556db2e7af5cdeec144ba06b68f757eb0", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.02734375, -0.955078125, 0.015625, 0.8046875], "student_probs": [0.08961271494626999, 0.09632836282253265, 0.25428760051727295, 0.5597713589668274], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4096f5003d1f3b129cf9243db6a1077a8d041855c9fe1701d9708f1c5cc6eb99:action", "state_id": "b1bd88d80ff01380bf7ea2ac5918aff59598a71c431bb8be03d9d82be2f81a1c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.02734375, -0.974609375, -0.01953125, 0.787109375], "student_probs": [0.09147901087999344, 0.09643255919218063, 0.25061601400375366, 0.5614724159240723], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "83bdfbdb58521e931075034e9d63a376c7f474a6775fb136c15bd39be074c290:action", "state_id": "4a7b2bff2839a401c7a7dbb04e95c4afe3f90f4cd4c0480f86255b38f0cd5987", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.001953125, -0.990234375, -0.009765625, 0.7890625], "student_probs": [0.09341906011104584, 0.09452025592327118, 0.25196316838264465, 0.5600975751876831], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6835077539c7adc41adf2dad38b0970c4a2130df875c7ed0acef0cb2d103e2dd:action", "state_id": "5d2998502cd69984b806a5ebfe531d48bcfbab8632721aad669df5683d68cb08", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01953125, -0.833984375, -0.03515625, 0.794921875], "student_probs": [0.09076277166604996, 0.10926717519760132, 0.2428937554359436, 0.5570763349533081], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9d24460705bd2f71afe88bb4da2e3ffd0f1c36506eac36ac6ae1f8369f3a1a3e:action", "state_id": "85c2c5340311123ef618d8d20ce97ecab54c8b4e7d7ac2ed50e4220fd311a62f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.6328125, -1.7421875, -1.52734375, 0.47265625], "student_probs": [0.08913787454366684, 0.07990267127752304, 0.0990528017282486, 0.7319067120552063], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7117c9357f031b76f0be6285a1f77de7bf5bdb34d4f15fb48d8ff997c6e7deb3:action", "state_id": "3b2edd4a2017f055a810526b5609d7dde7c6c6ced5fae842fe471bc6efe26de2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3515625, -1.21484375, -0.7822265625, 1.111328125], "student_probs": [0.06388839334249496, 0.07324840873479843, 0.11289674788713455, 0.7499663829803467], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "1fc74ff66f27fb5b6c107316efbb36f8ee9b9a7328feb1394181b8e6eea588c6:action", "state_id": "dc97f4f04cda735ceb73f712c068dc7dd03a162a3fbf342f3b38be1784e7921e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37109375, -1.27734375, -0.857421875, 0.91015625], "student_probs": [0.07375386357307434, 0.08100277185440063, 0.12327346950769424, 0.7219698429107666], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3dce5a48d28ebaaa91e003a8d4347255acadc7abf5bd80f39fa55183646aebc4:action", "state_id": "73adb6aa7b67acd2a3e1b04d8f6ebb96c6c702ebcbc651596102509c2f2c58ed", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.38671875, -1.15625, -0.7666015625, 0.521484375], "student_probs": [0.09208695590496063, 0.11595498025417328, 0.17120307683944702, 0.6207549571990967], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "22391d2b97279c056c9e0f0fe1f29fa7efc85241cce78880448d476c6792ffa1:action", "state_id": "0cb4eb86d3f99c2100f7b41fa5c2c62ac991de6b8a8ea7f3d1c5137d6301257e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.09375, -0.966796875, -0.173828125, 0.796875], "student_probs": [0.08874716609716415, 0.1007603257894516, 0.22267502546310425, 0.5878174901008606], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f5824c3be1a39677c1173a19edbd299e614796e1d0f8c9a080dbef757455655d:action", "state_id": "3b7ef41440fbd705c6415d0f3fbb6cf9b0371f503c4f38c008563caa40ab372c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.998046875, -0.970703125, -0.08203125, 0.783203125], "student_probs": [0.09556294977664948, 0.09821204841136932, 0.23884165287017822, 0.567383348941803], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "36b38fddcf934a4262612563c6ca59f9ef49ed66b9dd26767926e776f68297cf:action", "state_id": "9bc61b746a946758b104dc6b6104a77dc1473cde754cd1e630baab0b405c60a3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.12890625, -1.41796875, -0.234375, 0.591796875], "student_probs": [0.10221175849437714, 0.07655306160449982, 0.25002923607826233, 0.5712059140205383], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "17376ef0ce12555df120bda9800078b12817c9e03de9f672415ed65be111ac93:action", "state_id": "200b404ed4393f40ec5a3682a66921c60676b3cc01c5291505d1b8c7371a6658", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.984375, -1.056640625, -0.041015625, 0.8515625], "student_probs": [0.09285145252943039, 0.08637819439172745, 0.23849785327911377, 0.5822724103927612], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c49b53815188191de0182f06307fbd801995b7f5f173e3521a568e5345aa51d7:action", "state_id": "85aac37376f995fdb9672db0e7cd6f5ea33f19dc31b40058c9008e41c175690e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.06640625, -1.0625, -0.05078125, 0.78515625], "student_probs": [0.08980879932641983, 0.0901602953672409, 0.24797004461288452, 0.5720608830451965], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "acff25f069a51ef44b42abd9e9829060f4ab1d116123e076fe6ebf1a1cf3cb5f:action", "state_id": "6af7240a4c9164d9530e44a431fc9cb68885e0064c2c55229152d198ca8651c4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0546875, -0.96875, 0.001953125, 0.8203125], "student_probs": [0.08705305308103561, 0.09486503154039383, 0.25042471289634705, 0.5676571726799011], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c0f79f50b8f7132e3d103d600b8cc75faeec2ecd9687e35bf691868e14a39d81:action", "state_id": "8329e7e81d69f7a3dc34f1b9671ef8eef21f7aa382ad4a540a39f40f7d9d93c8", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.9375, 0.01171875, 0.83203125], "student_probs": [0.0877111628651619, 0.09652019292116165, 0.2493782788515091, 0.566390335559845], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ec7b0a80c2a1567a6461522075f543cc3b5aa0f7759b963da81a3c8ca72e98be:action", "state_id": "b87a9b60feb9196aeb410401440a5f89648f3a9a3e2b0658fc2dabb09abca48e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.939453125, 0.029296875, 0.837890625], "student_probs": [0.08705282211303711, 0.09560882300138474, 0.2518956959247589, 0.5654426217079163], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7f08acd1dec5fc0907042b33fb87834ccfbd30d1af1715c708f9470c7a798193:action", "state_id": "13d95288e2576dfd886a65c815d8cad6429dd9ecb0daa3c76037a22f5ea4b96b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.00390625, -0.921875, 0.033203125, 0.84765625], "student_probs": [0.08868153393268585, 0.09626289457082748, 0.2501750886440277, 0.5648804903030396], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "835f71add830811395ab18799d79137d87b04b023cd742437312f78e48aeb32e:action", "state_id": "2b97c7813bd755aa04bd991f2c6b5e2d81da891c48cb0516bba9ead8ab2e61c7", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.01171875, -0.921875, 0.025390625, 0.85546875], "student_probs": [0.08783388882875443, 0.09609057009220123, 0.2477838397026062, 0.5682917237281799], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ce07d3bc9866a16b8ceabf8766d4ad32b02eb19be869dc022b015be4cde39b22:action", "state_id": "0bb6c3c29991d7415e158f6c5f34a974f80801f3e7b8eaf3eb8af7d32282b637", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.025390625, -0.9453125, 0.005859375, 0.822265625], "student_probs": [0.08902442455291748, 0.09644654393196106, 0.2496751844882965, 0.564853847026825], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f39126ebc97f29966aa4855a61361cedf6a8a83e3b4d1b3476152828cbf1e7d4:action", "state_id": "c023e60b0c2a105a6d3be7aefebaae6286c19912b0edc0a6dd4d9455989d0b54", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.033203125, -0.9609375, 0.0, 0.806640625], "student_probs": [0.08944104611873627, 0.09614384174346924, 0.2513340413570404, 0.5630810260772705], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cda300b20fc839bd1151788f9440be58cd042fa0b4f1cf96aac4db19401202bb:action", "state_id": "a4a3c034ecc04b412541fda802b0a5b7e8571539e1028eaacf317dff8b1b414c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.037109375, -0.9609375, -0.021484375, 0.796875], "student_probs": [0.09009812027215958, 0.09722920507192612, 0.2487688958644867, 0.56390380859375], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "791bab846a8b898b31fcf53d1c6f9b9e3b52772bc1674c142394d77cea4a2421:action", "state_id": "1fef0a3095cb0b3751f3e2ed3ee4e6624977b9a87f07cba807f4673e079868fd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0546875, -0.990234375, -0.05078125, 0.732421875], "student_probs": [0.09287068247795105, 0.09905360639095306, 0.25343674421310425, 0.5546389818191528], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b05242b99c5e7eb2c1bcab2bc27650ba710e852eead6a244ee9d3e30e80d31e1:action", "state_id": "51c357d614bad17fe61f5f3e6e1d830582047735abf385079e53a29d41d59a5e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.048828125, -0.97265625, -0.048828125, 0.755859375], "student_probs": [0.09194978326559067, 0.0992274209856987, 0.24994540214538574, 0.5588773488998413], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "98956e2366ae577ffb97948a49735adaf5adddee9fa6428b0ffb977698fa5d67:action", "state_id": "8e24d2f9e72b4ad714e88b4479fc617328ee3b132b6e15b035fa3cc67b088013", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.06640625, -0.974609375, -0.076171875, 0.73046875], "student_probs": [0.09243186563253403, 0.10131847113370895, 0.24881413578987122, 0.5574355125427246], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "5e29a9554f3300213f90b4668091dd8f2a065a911357c118f16c97151276c065:action", "state_id": "1efb55242353250b8a2777785ab3ec9e6f9fb9bcbec52669378bf78a9ee01159", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.052734375, -0.97265625, -0.0078125, 0.732421875], "student_probs": [0.09185217320919037, 0.09951004385948181, 0.2611519396305084, 0.547485888004303], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c8ace3d08b3f38486a4576b7ae65ed99c4647a6b633fcabbe8c8df2131c3aea1:action", "state_id": "f5461d6492de1bf12c6a6eddc3431640c3dc2c48ead93c6d8b0b6aaf5c8f8249", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0703125, -0.9765625, 0.0078125, 0.771484375], "student_probs": [0.08814213424921036, 0.09680519998073578, 0.2590641677379608, 0.5559884905815125], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "8906b36f6e6b9955ffb4d039fdce0f49898cbdafd341a2cf35ff285f35debd72:action", "state_id": "c321726eb1dcc97fb5ed8c0c74d1f14b2847cb9b55237e98732348a63dcb567a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -0.958984375, 0.025390625, 0.7734375], "student_probs": [0.09002222865819931, 0.09752753376960754, 0.2609972059726715, 0.5514529943466187], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7b8adcce6e6be4806b3bef8bc6a75f7131583ab5defa7105211fde014be3a979:action", "state_id": "2461c60570ffdb6dd7f54542d524c07fb93f5bd5b34fa19745db5ee71d0ac1ce", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.046875, -0.966796875, 0.02734375, 0.7890625], "student_probs": [0.08863608539104462, 0.09602582454681396, 0.25950026512145996, 0.5558378100395203], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aff7bd9110b972680f7b4a32c286f6f2b8d35ac070f2a6731d1db26858dbce94:action", "state_id": "2dac763e2ddf98acdb0d0c2fc9875e799efae90d0aa3f0a533dadbc4712fd166", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.05078125, -0.96484375, 0.013671875, 0.779296875], "student_probs": [0.08909983187913895, 0.09709549695253372, 0.25832295417785645, 0.5554816722869873], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e0d41a119ddb681e422ce9f4b94c90a7c762d0fb2b764ec0f552300d61ea6fe4:action", "state_id": "0c5b1a7f44f32d6e469597a48a48d0b41c044be479bc3947e82f3bc3fc205658", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.046875, -0.966796875, 0.00390625, 0.787109375], "student_probs": [0.08926952630281448, 0.09671208262443542, 0.2553005516529083, 0.5587178468704224], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b52736d0b2c0e96d8648362420517185e9799b824c786e5bd0f6f6022dd5243a:action", "state_id": "9c49d432aa1b56d0437cceaa147c7ad2fc1fb4441ebf3114c733130d5fe4a2ce", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.056640625, -0.974609375, 0.005859375, 0.796875], "student_probs": [0.08801806718111038, 0.09554270654916763, 0.25468873977661133, 0.5617504715919495], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "49c142969843577cd2ad27858f0c352551fdd1d4ec38c487d6aa8e1cde639197:action", "state_id": "c20491280814090d7e40791e55d55cc258af6469d3636034de9e574d984af2e9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.072265625, -0.943359375, 0.005859375, 0.779296875], "student_probs": [0.0873628631234169, 0.09938254952430725, 0.2567737400531769, 0.556480884552002], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c675bbc52b0a30c6e62e1383da2b17a6d8aeb19505be9256e6bff48ec84289be:action", "state_id": "fefeb61007b8c166033cc3734514561831ec216652f9b630c1ca481bf88acbaa", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.06640625, -0.984375, 0.021484375, 0.802734375], "student_probs": [0.08668351173400879, 0.09409406036138535, 0.257277250289917, 0.5619451999664307], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "13b0b456b90cfc1da1d382e1a7e1c9f1d3fcca51e05f2818f350568b49e34239:action", "state_id": "0f092bbe18f90bf7a6d6c5e4dc218ad2e8d432b08a400983c95f4b7e4f77ab3e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.044921875, -0.966796875, 0.021484375, 0.828125], "student_probs": [0.08699983358383179, 0.09406924992799759, 0.2527276575565338, 0.566203236579895], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7596229994e2e25c6dc409737b1e39edb6bb9edce153be596e85daab9d7b688a:action", "state_id": "bb6de07b6474d768f4cc34d2785148bdc583fd7ba959fd25e1d14dcec182fbc2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.064453125, -1.001953125, 0.03125, 0.818359375], "student_probs": [0.08600122481584549, 0.09154782444238663, 0.2572541832923889, 0.5651968121528625], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c1a2dabd6b36154f208239e96ba99b2317dc93cf90fd49243980b5755332503f:action", "state_id": "38ab6e5a2f81d46fb60c706f67a30a1347f57770c482abbe5c80e82ee8bfacff", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -0.974609375, 0.060546875, 0.84375], "student_probs": [0.08574611693620682, 0.09163351356983185, 0.25799837708473206, 0.5646219849586487], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6e43a5fb3a90a988929a2e2ab2567fb289b56375a279beeabe05ace72ddedde1:action", "state_id": "85648d7e2b7f59821d5ce8e0d0de093cf89d37e4ca49e3a88adbb62080c2314d", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -0.9921875, 0.052734375, 0.8203125], "student_probs": [0.08704562485218048, 0.09158007055521011, 0.26037830114364624, 0.5609959363937378], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6f9825c348094b21991b2b2b9e8cc022f511438d611db5daeb73e16993055563:action", "state_id": "427463335e729b140cb8e41d7e4ee3011d24d21fc158015797ade5e65b9c9b50", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.05078125, -0.97265625, 0.01171875, 0.828125], "student_probs": [0.08679655194282532, 0.09384945034980774, 0.2511541545391083, 0.5681998133659363], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "930ba5b56d9dbbd053f487d169d6c08c882a69bbd7699bfca74ad16cfdcd4dcc:action", "state_id": "85d565d1084afb9b776467ed1167333a8242038000e309d801a7ab3586101145", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -0.97265625, 0.017578125, 0.828125], "student_probs": [0.08760076761245728, 0.09361550956964493, 0.2520003318786621, 0.5667834281921387], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a06a5a0c1efd6d03bcb8fd3e446b231fa7d99f747917ed8b8119ed4941aa0bcd:action", "state_id": "3b5f94227915229de79d0be3296d39dea058121cf081bc5ff87d3227681646cd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0234375, -1.0234375, 0.05078125, 0.8359375], "student_probs": [0.08812569081783295, 0.08812569081783295, 0.2580060064792633, 0.565742552280426], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "437492e47947d0f0e49f528d4e36840d0b8f1e60c4aca247a8e087070ceab618:action", "state_id": "e54074e208436b80c40e415b4f62d3113650309578ed1870aa3e7b304428c5ae", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.080078125, -0.966796875, 0.00390625, 0.70703125], "student_probs": [0.09050963819026947, 0.10136598348617554, 0.2675859332084656, 0.5405384302139282], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e19503ebb673ee219d3de05df6f02f2c55e6b029fcf1fade62c897004d07a0d2:action", "state_id": "926db535f0f4d0bc5292da1c05a8633902a27f8c5fe1bd93f300486dd9747e1c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.095703125, -1.390625, -0.3193359375, 0.90234375], "student_probs": [0.0885520800948143, 0.06593497097492218, 0.1924734115600586, 0.6530395746231079], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "75aa9906a7dcf6e30f10cb6c7f52036b342bc46f780177d5ca56b7c7af2d302d:action", "state_id": "01416f067d62fe745af779a3177840be29abcbdf07563b7637452e9efa9d0f17", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.63671875, -1.63671875, -1.4765625, 0.296875], "student_probs": [0.09912760555744171, 0.09912760555744171, 0.11634549498558044, 0.6853993535041809], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bd7c2d3f47654b5e9c6f314fba4a6cd478e18c055d652e4897eefc84a5f2c84e:action", "state_id": "3658474f0db8a027b1aaffe67f46af0aad06e44e1102f764894c731d4d040caf", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.734375, -1.828125, -1.37890625, 0.2734375], "student_probs": [0.09272680431604385, 0.08442871272563934, 0.13230717182159424, 0.6905373334884644], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2005fa82c8fc70b5851100e58eceb412b746787719d7e3b4696ed88e5568c42c:action", "state_id": "9e381c5b6664df5d298f477f2661aaef00d214e21a18ea1698db7c8bbba97d15", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.7265625, -1.5703125, -1.31640625, 0.01171875], "student_probs": [0.10679502785205841, 0.12485603988170624, 0.16094578802585602, 0.6074030995368958], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "66f158f420b9f016cf92e4c904f2e289ad2ebd33ab8a25e54ee9f471b87f0478:action", "state_id": "a5c83b4b54194a5c613919bc6f2d3157261a2c1839c958f7db15def25d9c3e61", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.703125, -1.61328125, -1.25390625, 0.203125], "student_probs": [0.0962565615773201, 0.10530499368906021, 0.15084244310855865, 0.6475960612297058], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b5e934431098149c41981b9a6da4bd3e5fa4c4ee254dfb61297b15097f6eca7c:action", "state_id": "6b9af706efafc74471c0eb713974c920e009cb9c5b7c11aaed7ef706fd717056", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.421875, -1.46484375, -0.98828125, 0.501953125], "student_probs": [0.09663840383291245, 0.09257392585277557, 0.14909295737743378, 0.6616947650909424], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d3c41c2fe934c6f71474e45fc50ccec16aae96ca16188ba5ffc05423efbc5260:action", "state_id": "8662e077ba1137a085ce6b4e20644fe904440a6c7263fabe1f403e7921cfbadd", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.072265625, -1.11328125, -0.4609375, 0.814453125], "student_probs": [0.09615036845207214, 0.09228648990392685, 0.17719334363937378, 0.6343697905540466], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "555fcd8e9e6db4c2cc1cb879218f8540516c983166a68232fdee4f5b28b46dc2:action", "state_id": "eb2dfdcf3a7973546ab11be9d0245d96436dd595768f1c4c5acb7a54e8c4f07e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.16015625, -1.087890625, -0.4619140625, 1.0859375], "student_probs": [0.07387758791446686, 0.07941403985023499, 0.1485099196434021, 0.6981983780860901], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c78000c46319b0e8c8970b4b8f307f9599189e68398cd5a928ba2b768029e5a3:action", "state_id": "c8ad172a4ca0b1dcaca043ab8555d172da376e42bea18e5db4a0866fca0590ed", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0859375, -1.2265625, -0.49853515625, 0.931640625], "student_probs": [0.08937977254390717, 0.07765448838472366, 0.16082176566123962, 0.6721439361572266], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4b546a42313699e402756de21fa3354db4cc913d346361e5fadf2255f8ed940e:action", "state_id": "6e5b9e705510ad56d75ddda3d5413af93d0813bdf65bdfbcf8b1370554932a9c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.171875, -1.13671875, -0.5009765625, 0.765625], "student_probs": [0.09146494418382645, 0.0947377011179924, 0.17890486121177673, 0.6348925232887268], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b46894dfe702f1c7bb2f37df0509ab39ea67737e0e4a7be5ca1cb511d5afbd28:action", "state_id": "3096d29bd8d8aff1dda989bedf3272214787eb6325528d0bf03549ccc609c64a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.34765625, -1.19140625, -0.6943359375, 0.6015625], "student_probs": [0.08997476100921631, 0.10519114136695862, 0.1729235202074051, 0.6319105625152588], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a71861db6789dda7d8e7f064537f061eb881f06b6e4ac83aa38641d8d65b51a0:action", "state_id": "214b1150a629d952bed3702d8f4d59909fe73d2e4b0ee85c36a5bef37735c0d1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.140625, -1.0078125, -0.53759765625, 0.71875], "student_probs": [0.09625193476676941, 0.10992315411567688, 0.1759141981601715, 0.6179107427597046], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "dc5b0bf7b667ce6f2199bdba61c5bdb0f37e98d60604040cc109a4db0ef18b96:action", "state_id": "ddba2582c688e5a6c4831ae033158da790289e2b9babaded50d050069d10991e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.080078125, -0.947265625, -0.4130859375, 0.869140625], "student_probs": [0.08997983485460281, 0.1027601882815361, 0.1753138303756714, 0.6319462060928345], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c4d8b3155fc22a7af6594f3290d2ccd481de3c75d027342db8a6e505e6cba04a:action", "state_id": "6413a11993531aca822ec907a976e41657d34000573f75b5a41e56aa434bf663", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.087890625, -0.95703125, -0.42578125, 0.876953125], "student_probs": [0.08918630331754684, 0.10165522247552872, 0.17292135953903198, 0.6362370848655701], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4bc1138223e117f7bc0d5bd7319de19771eebc1e9454b5892ad42d0e3ce30a10:action", "state_id": "9cdb786a31a65156d3e6fe22fbc2688a43f6a4591afc68da7215e9212289e8d5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.07421875, -0.951171875, -0.3896484375, 0.92578125], "student_probs": [0.08693449199199677, 0.0983174741268158, 0.1723840981721878, 0.6423638463020325], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "99abea423e707dee693aba7a46fd467d5e304b86ef0d22c739382d60e75080e2:action", "state_id": "f6dd190669b49f11ee1d6b201a1653ab515f88541ac2d4db9e57dba9302e2181", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.080078125, -0.955078125, -0.37890625, 0.91015625], "student_probs": [0.08721048384904861, 0.09882242977619171, 0.1758262813091278, 0.6381407976150513], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b5bd17456aabb2fa5c60dde71dff3389380421af08622a7bb857dc5d0180de4d:action", "state_id": "f7922c29c64184c3eaf2954fa1cbfb775a95226b955ffe78610b8cdb875046fe", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.083984375, -0.95703125, -0.3408203125, 0.876953125], "student_probs": [0.08815290033817291, 0.10008560866117477, 0.18534831702709198, 0.6264132261276245], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "884012333a77284aa1420e19b83c9d75e5009597901ed075cfe4c92e305b4ae6:action", "state_id": "cd91a36d3c6625ed75181b5dbb581b46c8cd9267d6a23409adfed5be343f4f40", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.07421875, -0.958984375, -0.3212890625, 0.845703125], "student_probs": [0.09036874026060104, 0.1014060527086258, 0.19187194108963013, 0.6163532733917236], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "4be193d8826cbeb9e356c2cb60a75ed9c872e56054fdd78566085b205e21573c:action", "state_id": "b235955dbdde8f78299ff2cbcd30b20ad43b484f859f47de39c0c4573a6446a4", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1171875, -0.98046875, -0.4033203125, 0.78515625], "student_probs": [0.09182769805192947, 0.10528097301721573, 0.18750043213367462, 0.6153908967971802], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "ff76d1dc755b748f149c3b817544434cffdff9fbccea1cdba43eed38df3e0ec5:action", "state_id": "51f7352ef1068464a0fa218bc644b8cc4c1c7c8fa41a6164567d7aa9261bde91", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.10546875, -1.00390625, -0.3701171875, 0.76953125], "student_probs": [0.09333640336990356, 0.10331396758556366, 0.1947198212146759, 0.6086297631263733], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "231fcb12611a828ff5e5712c5567f1165a4c25785971f997005deb3d37722460:action", "state_id": "6051eced32ac30eb1b476f29a913d3e323885b39ceb6c636bc2b2fba4f139208", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1796875, -1.037109375, -0.5823974609375, 0.8046875], "student_probs": [0.08892896771430969, 0.10255672037601471, 0.16160061955451965, 0.6469137072563171], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c18b205fa41ad47616b03d9e594fd73bc2fe6f9c5668a37985819ed2e3004c9e:action", "state_id": "41cd3c8f939caee67222f13bc466005d0ec915118952dc082a87c6e6eb232c71", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.57421875, -1.359375, -1.109375, 0.53125], "student_probs": [0.0830400362610817, 0.1029420793056488, 0.13218025863170624, 0.6818376183509827], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aa37f23b247a69edac3873d8fc8266a086370254f7e8b4c552e4965ceeda539e:action", "state_id": "0af8451adf4a00827f311649b8309a70ee92413e111e8a859dc95fe676eef605", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.62109375, -1.4375, -1.140625, 0.435546875], "student_probs": [0.08592530339956284, 0.10324162244796753, 0.13892678916454315, 0.6719063520431519], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bf049691ed4653921f920ead7781b0fd6ba647df3af7586cbb9142795a0424a7:action", "state_id": "cfdb14c92bb04cc847e0b0a777667ee4dcd7ffd66e0d4c4fc17f81733d70dc25", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5859375, -1.44921875, -1.2421875, 0.60546875], "student_probs": [0.079971082508564, 0.09168729931116104, 0.11277730762958527, 0.7155642509460449], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f8f88e88705b33b6131dbd7274ae5eceb049c88cd0bddbc3856eee9e6f6a2180:action", "state_id": "df292ac5d1fc7ab72187af92cc4a5e5f39312b0062e64b5880e7681f01e6fb88", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.18359375, -1.47265625, -0.45263671875, 0.484375], "student_probs": [0.10955996066331863, 0.08205662667751312, 0.22756342589855194, 0.5808199048042297], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7fdb48b6ea264a95ecc0a79af70759f42aab0ffb02d38f0fab5c04b6cca486ca:action", "state_id": "b6937e6006f48907081c5e113e6b8c6b0775cb690826df873dfee150717fd29f", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.10546875, -1.126953125, -0.3212890625, 0.798828125], "student_probs": [0.09187766164541245, 0.08992478251457214, 0.2012680470943451, 0.6169295310974121], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b3cdeab3df1c75fbc8907fa74536b69668d797ce7d7aca1fc7fccd104cef2293:action", "state_id": "40d4df5dfd563cea7de22292fa22a85031447d272fa0f93e0987cec6e3fc8227", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.17578125, -1.23828125, -0.6572265625, 0.7109375], "student_probs": [0.09787900000810623, 0.09194882214069366, 0.16439741849899292, 0.645774781703949], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "554e00a98123f61f0dcfd1507b89702821f6dae4c7a1d25a9142cf9134c7379a:action", "state_id": "a9df0f3818f63c8c7fbfd9bee9a171517454d7a4a4b437290a290ffe34fbf5da", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.111328125, -1.25, -0.531005859375, 0.751953125], "student_probs": [0.09899052232503891, 0.08617260307073593, 0.1768578737974167, 0.6379790306091309], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "9f119bc7f58a899de0f986ba9af6c1244b469db479ec664bb3420bd1f9687069:action", "state_id": "87f5055c7a5b14bc9c0af2af6671d8afee1ffa4b876a43cc1ab322142b6a7019", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.125, -1.21484375, -0.65625, 1.123046875], "student_probs": [0.07703392952680588, 0.0704147145152092, 0.12309987097978592, 0.7294514775276184], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "728e9d8e6cc7a58455e00d43a707496532264e0c4305f23d4bd3add625031ecd:action", "state_id": "3ca33f70fa432dfc19dfdc4bbb89a6e9db1a3b307fe370b7db82c162828ab2a9", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.43359375, -1.53515625, -0.892578125, 0.724609375], "student_probs": [0.0814548060297966, 0.0735882818698883, 0.13991904258728027, 0.7050378322601318], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bdc6952adba2c9170309941d347ead7c8bbe62c5188ec51e2e6afddffd3e656d:action", "state_id": "307de83f57efd13d7808b4eaae4ba356f2a180554e9ed3107f2cc1c5b93bb36b", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9453125, -0.953125, -0.1171875, 1.2734375], "student_probs": [0.0742001086473465, 0.0736226812005043, 0.16984574496746063, 0.6823315024375916], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7a5684164ce4a83b0f570bf63c700e371c75da7e7c3c4c06cfd08e21ac91477f:action", "state_id": "460e944201f32324682afb64786f0b5de965bda31d1514c1d3d24edbd36796ff", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.9375, -0.98046875, -0.11328125, 1.314453125], "student_probs": [0.0727572962641716, 0.06969722360372543, 0.1658938229084015, 0.6916515827178955], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "0ea715c33a11504facbcf31c91609dc0ec986f89ea4a149d3c9670d6813bea60:action", "state_id": "911deda7913afcefac578df9839bce90be5189ef0c1abed05c5e67dc3c0c74a6", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.990234375, -1.041015625, -0.205078125, 0.9609375], "student_probs": [0.0894441083073616, 0.0850154310464859, 0.19612853229045868, 0.6294118762016296], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f05ce608a78ebbeda45568006eff2ef7dcb8922fa82234545f5af024ac0b3065:action", "state_id": "82830c00d94be5d5523a2c666ba07b9b31a4b4cc096ff68d60b66d5a36db3478", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.15625, -1.1171875, -0.390625, 0.908203125], "student_probs": [0.08284208178520203, 0.08614213019609451, 0.17813846468925476, 0.6528773307800293], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "3604a690146eb95c4c1a22361052f2c14f918021bbe8859db0db46c47a1d0eaa:action", "state_id": "fc3b6371950b67314d573128442ee0535ab4e9df9100bcbd5d9251e25f71e67c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.00390625, -1.19921875, -0.2939453125, 0.92578125], "student_probs": [0.09307652711868286, 0.07656266540288925, 0.18930946290493011, 0.6410513520240784], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "2ba12f6a4692ee04627e65ad42fcd068c981692690b805f3efba1479ddfd87f5:action", "state_id": "045aa6d5e4267f7a202e184581236e214b2da97191cf31707cc02aa5cabcdc83", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.24609375, -1.1875, -0.479736328125, 0.59765625], "student_probs": [0.09494464099407196, 0.10067402571439743, 0.20431265234947205, 0.6000686287879944], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "bb79ebf74bc805752a2318ebf0fe6527651963fbd2d145760b29ac27bfebd661:action", "state_id": "bda074e787cb5bc8f86d6860d73ec1a9193e119cc6885fa08be0a0db61bf62e2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.34765625, -1.234375, -0.575927734375, 0.55078125], "student_probs": [0.0912499874830246, 0.102195143699646, 0.19741958379745483, 0.6091352701187134], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f85b9798476729149557ea1125b01076e9f3ab37812554ad32c847fe354baa91:action", "state_id": "24d0822d5f9f921410d229c6bbee19bb3ab702ccb892f6bfd72c75d5606eafc2", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.15625, -1.046875, -0.31640625, 0.70703125], "student_probs": [0.09194189310073853, 0.1025685966014862, 0.21293790638446808, 0.5925516486167908], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "88fd551a2012ea6e072b5e805c91f2ec857d5e7ef6d0d4dd1682cb432698bb8b:action", "state_id": "39fbd4e089c44b8aeeaf652e308846e14278888fa1416c55b51a02a3b6265043", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.037109375, -0.978515625, -0.171875, 0.806640625], "student_probs": [0.09297047555446625, 0.0985807254910469, 0.22085721790790558, 0.5875915288925171], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "af3198153a8dd2459cc55b9879545b6003ccdcacbab2ff0be5ffd574a5688d9f:action", "state_id": "20e23f834454121bfa052f014b84c96c633405985a46faa45873a53a724e49ef", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.041015625, -0.986328125, -0.1640625, 0.8046875], "student_probs": [0.09265843033790588, 0.09786681085824966, 0.22271057963371277, 0.5867642164230347], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "59d6c2fc1eb34dd0475d50c62b9518d5f0ab213ca00d0bfa4c62679c7ae4541e:action", "state_id": "109ad661e6f80f55d22677ae6181e28c952c7f6ae83c60c3b0b6980ec38b01f1", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.04296875, -0.990234375, -0.10546875, 0.8046875], "student_probs": [0.09130193293094635, 0.0962458923459053, 0.23314765095710754, 0.5793044567108154], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "497c7b81b84b637faa79e8b308e3684b337c9a6d078d93398270508a595faebd:action", "state_id": "b2b7172b6989a738bbad03e0c16b5994ef865af3dd919ce4aa4c15a9a53b462e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.052734375, -0.994140625, -0.1328125, 0.787109375], "student_probs": [0.09203869849443436, 0.09759272634983063, 0.2309338003396988, 0.5794346928596497], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "7f0bc64bde1fecdfd4673c12fa371ce3acb6d60d5f2b73ed7dfeef259825bdc7:action", "state_id": "462213461d2038bf85f28da76c70a314c60d695ed307d16fe4540a22ba4fd36c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.07421875, -0.994140625, -0.18359375, 0.779296875], "student_probs": [0.09172425419092178, 0.09937146306037903, 0.2235001027584076, 0.5854041576385498], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f15f9dc8b82f5d84427421e59458ed30aa4bcd89e396ab5f526c1ab8f624fe93:action", "state_id": "544f2208b4fc9f5d084759e7aae8ed7c9751bd8b86e43ae1aadbcc2ce346aa91", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0625, -0.966796875, -0.287109375, 0.861328125], "student_probs": [0.08993619680404663, 0.09896869957447052, 0.19529108703136444, 0.6158039569854736], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "373374bfcc4df16fa4549eac31119b672de155cea3adbb8fcd961662a17fa780:action", "state_id": "bccb9d58f953f546c2d00d770458f6993616e697e58f49b4174c91650f7d548c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.06640625, -0.96875, -0.3125, 0.876953125], "student_probs": [0.0892057716846466, 0.09835683554410934, 0.1895877569913864, 0.6228496432304382], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "772c55bc43e080ecbc7bd56041df5a602a3dac7f6e6c29151b4d12dc1dfb5e29:action", "state_id": "9ba07f82f77f552a7d686a2ce5745562baaa781eadbc0ad87d2b155a980aa722", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.095703125, -0.986328125, -0.3642578125, 0.8515625], "student_probs": [0.08926054835319519, 0.09957734495401382, 0.18549074232578278, 0.62567138671875], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "e7af888241a93292adbee3f1f580f7abb8d0fbecb2fe8f6f167f68b6ca2e410a:action", "state_id": "c2b73192b94602c9e97b93e3bbd78ddb89b25a43e20ec572f07f7f8e4fd44d55", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.1328125, -0.990234375, -0.3935546875, 0.84765625], "student_probs": [0.08700371533632278, 0.10033644735813141, 0.18221889436244965, 0.6304410099983215], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "71e7de38432cbac3a2f94e1c9085383e962f8ae680d77a75ce2bdb6bd56f3756:action", "state_id": "607711eda14ed621af52fd2d64d0b9584d38ccec598aef0fdf6fda51934ef98e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.123046875, -0.9765625, -0.4072265625, 0.880859375], "student_probs": [0.08604669570922852, 0.09962114691734314, 0.17603985965251923, 0.6382923126220703], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d8ad186c9d76034a44261fc831a322a15ff3981b782b1fa8ffb291a0330234e5:action", "state_id": "12ea28ed28f1fa2e6959126f083911fb5ca035fa034b95a5025c02fe3d5c1829", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.154296875, -1.001953125, -0.48974609375, 0.830078125], "student_probs": [0.08785279840230942, 0.10230989754199982, 0.17075221240520477, 0.6390851140022278], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "77ce51cc9d798ba22bf1c7cc08a87d8ca35df736634aea0532e57ce4c549c72b:action", "state_id": "a5c9ac734c53253267dd183492cb39aed2c168cbdbda627306bf187705ed26ae", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.16796875, -1.0078125, -0.567138671875, 0.892578125], "student_probs": [0.08440537005662918, 0.09906609356403351, 0.15392431616783142, 0.6626042127609253], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a4a05a411146847304eae412477be8aca06d358345f768dda50174dc443c15ef:action", "state_id": "1c1c22455538a05889807934725b6126399b4ab71e232958fd085c1681b24e6c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.55078125, -1.36328125, -1.0625, 0.587890625], "student_probs": [0.08114313334226608, 0.09787730872631073, 0.13222381472587585, 0.6887556910514832], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "67822a2751f8ff4a01e787a322b388da0709d93f17184a57d6414a53bd126f04:action", "state_id": "a2a572558396dbff1c28185dfe447299d4fe1a20854b14fef73b334221e1d09e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.60546875, -1.41015625, -1.08203125, 0.49609375], "student_probs": [0.08276467025279999, 0.10061624646186829, 0.13969182968139648, 0.6769272089004517], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "87db5c1d52a008d86980f7b7bbf327f2b12d914b30f223baa42475e61d77f8ee:action", "state_id": "7cc0ed3cbae7418e6fb2dee149ff770cc1f792e1335563b690fa994ed72ed4b3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.40234375, -1.30078125, -0.95703125, 0.912109375], "student_probs": [0.07253070175647736, 0.08028417080640793, 0.11321882158517838, 0.7339662909507751], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "277b6e301c7562e5e0ad548b8ce8b4600407d5e5e8400a19599aa70e30b79d67:action", "state_id": "c82be5cbb50b474b56ca509f537653163a87cfcde8a29c2ece7a30dc88b8d529", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37109375, -1.33203125, -0.94140625, 0.9765625], "student_probs": [0.07123707234859467, 0.07407483458518982, 0.1094755083322525, 0.7452126145362854], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "cd182f7d573b4c86c1e734a1e30a09956c453148315fa0a6d8eb649023f03f3e:action", "state_id": "c271e131243c4c0ba8788529dee7bcee92f07d87ce7b09d4a5f974bd797669d5", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.38671875, -1.33984375, -0.96484375, 0.990234375], "student_probs": [0.06971146911382675, 0.07305699586868286, 0.10629729926586151, 0.7509341835975647], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "f900df0be8f4e3e7877d543299097faa908528b6c651047dd1a40cddc9d3749f:action", "state_id": "ae088c500b19940c70fc0fa677318c59863dd0eb4f1f595834419b87d22da2ed", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.359375, -1.3359375, -0.939453125, 0.982421875], "student_probs": [0.07170790433883667, 0.07340840995311737, 0.10912815481424332, 0.7457555532455444], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "d5157efae8d396bf716b701ded7eaccd442f99fce5fa9a4a70a22bec2bc3bea9:action", "state_id": "03869c77f2db001075ccdd2198f0094bb969b575d4fec1fa71fdf6f598af55ac", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3828125, -1.28125, -0.9765625, 0.9765625], "student_probs": [0.07046018540859222, 0.07799231261014938, 0.10577327013015747, 0.7457741498947144], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "df5cc3d38b9de84bcf5bc8ec5f85f50f718d718d50762a3e4fc0da2cd0c701a7:action", "state_id": "44bcf34c3ae3401538cd76ff76822d9b21582055b36351d832a9ee0eec582327", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.3828125, -1.3046875, -0.986328125, 0.9921875], "student_probs": [0.06983795762062073, 0.0755128413438797, 0.10382036119699478, 0.7508288025856018], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b9b557b9901e747c8dd113df70fa460c927d268cac35bced6d3688e1ced86283:action", "state_id": "c67727655845cb085282d718f3a414b032a6758833b461a413dba513818ffe3a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.37890625, -1.32421875, -0.9765625, 0.984375], "student_probs": [0.0705353170633316, 0.07450014352798462, 0.10547324270009995, 0.7494913339614868], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "a4cec942f92faf8f9bad0b4c35683f00907355932a70fb84392a5227cc57a317:action", "state_id": "615f062ff99aa2ea3dd63fb6f02da43a6d6190010241e8841dd7e4b7fb710e26", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.44921875, -1.31640625, -1.109375, 0.8359375], "student_probs": [0.07477202266454697, 0.08539232611656189, 0.10503435879945755, 0.7348012328147888], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "b45405a38d36b7c505fdec377ca8f6c35567bd80618ac2ef03b64f960e994865:action", "state_id": "3302177b2ca0394502d5c08d0ea47a894c955d55cdde1e24f1a113c95f06903a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 0, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.5703125, -1.41015625, -1.1953125, 0.64453125], "student_probs": [0.07819425314664841, 0.09177614003419876, 0.1137719601392746, 0.716257631778717], "gold_probs": null, "gold_distribution_probs": [1.0, 0.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [1.0, 0.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "6b7f9a1195c0827da4cb0aa00b12f542fc136ee0e415d64acbdebec0e567ecb8:action", "state_id": "35dacdbf1376d1bf855c5b2e7a3914558b7da7377dce1fad30d7a19e53c9c51a", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.107421875, -1.41015625, -0.212890625, 0.5703125], "student_probs": [0.10483941435813904, 0.07745486497879028, 0.25645700097084045, 0.5612487196922302], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "718604e955caffd31ba8c30101ef9132c2d15c1e5f3f391c352a557456f91f34:action", "state_id": "d32142df993378b0c51a977d627dbfa40aa53cf435d651a8994fade45373448e", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 1, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-0.982421875, -1.041015625, -0.072265625, 0.845703125], "student_probs": [0.09389662742614746, 0.08855295181274414, 0.23330596089363098, 0.584244430065155], "gold_probs": null, "gold_distribution_probs": [0.0, 1.0, 0.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 1.0, 0.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "c6cc3a5218dea31bf3333346fdda79c35d14c46b3264d2eedd2ba7fc3bac8f42:action", "state_id": "bfce3eaffb800ce9f0acc89c8dcd8489f737f934d0672ac0fd783b2c209045f3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 3, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.056640625, -1.044921875, -0.0546875, 0.78515625], "student_probs": [0.09055308252573013, 0.09162048995494843, 0.2466300129890442, 0.5711963772773743], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 0.0, 1.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 0.0, 1.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "75be1dfe4243d8fa67269ef29e229cbcd9dd89ab78bf2ad5a84fb3f7d698b7a1:action", "state_id": "39fe4603b7343a4a529889dc3fb789c32723de0ae9a381569e462e3bb99b8b8c", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0703125, -0.95703125, 0.001953125, 0.802734375], "student_probs": [0.0865798220038414, 0.09696480631828308, 0.25298556685447693, 0.5634697675704956], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"} {"id": "aaa6ca92484bdf51c0f05e4335d2400616467ead6cc72f1d5b4fe0dd0c85c20e:action", "state_id": "c8a1360457acb848e4a18bd6ad69f1ecae2897834b8f6139303f2a02ab2943c3", "family_id": "unified_shooting", "split": "dev", "qid": "action", "type": "choice", "candidate_ids": ["left", "noop", "right", "shoot"], "gold_index": 2, "teacher_raw_probs": null, "teacher_probs": null, "teacher_rounding": null, "teacher_target_kind": null, "student_logits": [-1.0390625, -0.837890625, 0.044921875, 0.818359375], "student_probs": [0.0863075852394104, 0.10553992539644241, 0.2551628053188324, 0.552989661693573], "gold_probs": null, "gold_distribution_probs": [0.0, 0.0, 1.0, 0.0], "gold_probs_kind": null, "gold_label_kind": "reference_argmax_compatibility", "teacher_target_error": null, "task": "shooting", "record_role": "policy", "continuation_policy_id": "12aa0296306df4feafe25394ec070720a9da52adb612d2060d43ef3cc7dcad35", "training_target": [0.0, 0.0, 1.0, 0.0], "target_objective": "gold_distribution", "policy_target_kind": "expert_action", "scenario": "predict_position", "target_transform": "expert_action_one_hot"}