diff --git "a/NLLB_Decoder_1024_step.mlmodelc/model.mil" "b/NLLB_Decoder_1024_step.mlmodelc/model.mil" new file mode 100644--- /dev/null +++ "b/NLLB_Decoder_1024_step.mlmodelc/model.mil" @@ -0,0 +1,1956 @@ +program(1.0) +[buildInfo = dict, tensor>({{"coremlc-component-MIL", "3520.4.1"}, {"coremlc-version", "3520.5.1"}})] +{ + func main(tensor encoder_attention_mask, tensor encoder_hidden_states, tensor input_ids, tensor past_cross_key, tensor past_cross_value, tensor past_self_key, tensor past_self_value) [FlexibleShapeInformation = tuple, dict, tensor>>, tuple, dict, list, ?>>>>((("DefaultShapes", {{"encoder_attention_mask", [1, 1]}, {"encoder_hidden_states", [1, 1, 1024]}, {"past_cross_key", [12, 1, 16, 1, 64]}, {"past_cross_value", [12, 1, 16, 1, 64]}, {"past_self_key", [12, 1, 16, 1, 64]}, {"past_self_value", [12, 1, 16, 1, 64]}}), ("RangeDims", {{"encoder_attention_mask", [[1, 1], [1, 1024]]}, {"encoder_hidden_states", [[1, 1], [1, 1024], [1024, 1024]]}, {"past_cross_key", [[12, 12], [1, 1], [16, 16], [1, 1024], [64, 64]]}, {"past_cross_value", [[12, 12], [1, 1], [16, 16], [1, 1024], [64, 64]]}, {"past_self_key", [[12, 12], [1, 1], [16, 16], [1, 1023], [64, 64]]}, {"past_self_value", [[12, 12], [1, 1], [16, 16], [1, 1023], [64, 64]]}})))] { + tensor cast_0_dtype_0 = const()[name = tensor("cast_0_dtype_0"), val = tensor("fp32")]; + tensor cast_1_dtype_0 = const()[name = tensor("cast_1_dtype_0"), val = tensor("fp32")]; + tensor cast_2_dtype_0 = const()[name = tensor("cast_2_dtype_0"), val = tensor("fp32")]; + tensor cast_3_dtype_0 = const()[name = tensor("cast_3_dtype_0"), val = tensor("fp32")]; + tensor decoder_embed_tokens_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(64))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(262355072))), name = tensor("decoder_embed_tokens_weight_palettized"), shape = tensor([256206, 1024])]; + tensor decoder_embed_positions_weights_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(262356160))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263406848))), name = tensor("decoder_embed_positions_weights_palettized"), shape = tensor([1026, 1024])]; + tensor decoder_layers_0_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_0_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263407936)))]; + tensor decoder_layers_0_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_0_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263412096)))]; + tensor decoder_layers_0_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_0_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263416256)))]; + tensor decoder_layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263420416))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(264469056))), name = tensor("decoder_layers_0_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_0_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_0_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(264470144)))]; + tensor decoder_layers_0_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(264474304))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(265522944))), name = tensor("decoder_layers_0_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_0_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_0_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(265524032)))]; + tensor decoder_layers_0_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(265528192))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(266576832))), name = tensor("decoder_layers_0_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_0_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_0_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(266577920)))]; + tensor decoder_layers_0_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(266582080))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(267630720))), name = tensor("decoder_layers_0_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_0_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_0_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(267631808)))]; + tensor decoder_layers_0_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_0_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(267635968)))]; + tensor decoder_layers_0_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_0_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(267640128)))]; + tensor decoder_layers_0_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(267644288))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(268692928))), name = tensor("decoder_layers_0_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_0_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_0_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(268694016)))]; + tensor decoder_layers_0_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(268698176))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(269746816))), name = tensor("decoder_layers_0_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_0_final_layer_norm_bias = const()[name = tensor("decoder_layers_0_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(269747904)))]; + tensor decoder_layers_0_final_layer_norm_weight = const()[name = tensor("decoder_layers_0_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(269752064)))]; + tensor decoder_layers_0_fc1_bias = const()[name = tensor("decoder_layers_0_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(269756224)))]; + tensor decoder_layers_0_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(269772672))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(273967040))), name = tensor("decoder_layers_0_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_0_fc2_bias = const()[name = tensor("decoder_layers_0_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(273968128)))]; + tensor decoder_layers_0_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(273972288))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(278166656))), name = tensor("decoder_layers_0_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_1_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_1_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(278167744)))]; + tensor decoder_layers_1_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_1_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(278171904)))]; + tensor decoder_layers_1_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_1_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(278176064)))]; + tensor decoder_layers_1_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(278180224))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(279228864))), name = tensor("decoder_layers_1_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_1_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_1_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(279229952)))]; + tensor decoder_layers_1_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(279234112))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(280282752))), name = tensor("decoder_layers_1_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_1_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_1_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(280283840)))]; + tensor decoder_layers_1_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(280288000))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(281336640))), name = tensor("decoder_layers_1_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_1_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_1_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(281337728)))]; + tensor decoder_layers_1_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(281341888))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(282390528))), name = tensor("decoder_layers_1_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_1_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_1_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(282391616)))]; + tensor decoder_layers_1_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_1_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(282395776)))]; + tensor decoder_layers_1_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_1_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(282399936)))]; + tensor decoder_layers_1_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(282404096))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(283452736))), name = tensor("decoder_layers_1_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_1_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_1_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(283453824)))]; + tensor decoder_layers_1_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(283457984))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(282390528))), name = tensor("decoder_layers_1_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_1_final_layer_norm_bias = const()[name = tensor("decoder_layers_1_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(284506624)))]; + tensor decoder_layers_1_final_layer_norm_weight = const()[name = tensor("decoder_layers_1_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(284510784)))]; + tensor decoder_layers_1_fc1_bias = const()[name = tensor("decoder_layers_1_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(284514944)))]; + tensor decoder_layers_1_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(284531392))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(282390528))), name = tensor("decoder_layers_1_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_1_fc2_bias = const()[name = tensor("decoder_layers_1_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(288725760)))]; + tensor decoder_layers_1_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(288729920))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(292924288))), name = tensor("decoder_layers_1_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_2_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_2_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(292925376)))]; + tensor decoder_layers_2_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_2_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(292929536)))]; + tensor decoder_layers_2_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_2_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(292933696)))]; + tensor decoder_layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(292937856))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(293986496))), name = tensor("decoder_layers_2_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_2_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_2_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(293987584)))]; + tensor decoder_layers_2_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(293991744))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(295040384))), name = tensor("decoder_layers_2_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_2_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_2_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(295041472)))]; + tensor decoder_layers_2_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(295045632))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(296094272))), name = tensor("decoder_layers_2_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_2_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_2_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(296095360)))]; + tensor decoder_layers_2_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(296099520))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(297148160))), name = tensor("decoder_layers_2_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_2_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_2_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(297149248)))]; + tensor decoder_layers_2_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_2_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(297153408)))]; + tensor decoder_layers_2_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_2_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(297157568)))]; + tensor decoder_layers_2_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(297161728))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(298210368))), name = tensor("decoder_layers_2_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_2_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_2_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(298211456)))]; + tensor decoder_layers_2_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(298215616))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(299264256))), name = tensor("decoder_layers_2_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_2_final_layer_norm_bias = const()[name = tensor("decoder_layers_2_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(299265344)))]; + tensor decoder_layers_2_final_layer_norm_weight = const()[name = tensor("decoder_layers_2_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(299269504)))]; + tensor decoder_layers_2_fc1_bias = const()[name = tensor("decoder_layers_2_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(299273664)))]; + tensor decoder_layers_2_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(299290112))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(303484480))), name = tensor("decoder_layers_2_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_2_fc2_bias = const()[name = tensor("decoder_layers_2_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(303485568)))]; + tensor decoder_layers_2_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(303489728))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(307684096))), name = tensor("decoder_layers_2_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_3_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_3_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(307685184)))]; + tensor decoder_layers_3_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_3_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(307689344)))]; + tensor decoder_layers_3_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_3_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(307693504)))]; + tensor decoder_layers_3_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(307697664))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(308746304))), name = tensor("decoder_layers_3_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_3_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_3_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(308747392)))]; + tensor decoder_layers_3_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(308751552))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(309800192))), name = tensor("decoder_layers_3_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_3_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_3_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(309801280)))]; + tensor decoder_layers_3_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(309805440))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(310854080))), name = tensor("decoder_layers_3_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_3_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_3_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(310855168)))]; + tensor decoder_layers_3_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(310859328))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(311907968))), name = tensor("decoder_layers_3_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_3_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_3_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(311909056)))]; + tensor decoder_layers_3_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_3_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(311913216)))]; + tensor decoder_layers_3_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_3_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(311917376)))]; + tensor decoder_layers_3_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(311921536))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(312970176))), name = tensor("decoder_layers_3_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_3_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_3_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(312971264)))]; + tensor decoder_layers_3_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(312975424))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(314024064))), name = tensor("decoder_layers_3_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_3_final_layer_norm_bias = const()[name = tensor("decoder_layers_3_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(314025152)))]; + tensor decoder_layers_3_final_layer_norm_weight = const()[name = tensor("decoder_layers_3_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(314029312)))]; + tensor decoder_layers_3_fc1_bias = const()[name = tensor("decoder_layers_3_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(314033472)))]; + tensor decoder_layers_3_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(314049920))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(318244288))), name = tensor("decoder_layers_3_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_3_fc2_bias = const()[name = tensor("decoder_layers_3_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(318245376)))]; + tensor decoder_layers_3_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(318249536))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(322443904))), name = tensor("decoder_layers_3_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_4_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_4_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(322444992)))]; + tensor decoder_layers_4_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_4_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(322449152)))]; + tensor decoder_layers_4_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_4_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(322453312)))]; + tensor decoder_layers_4_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(322457472))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(323506112))), name = tensor("decoder_layers_4_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_4_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_4_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(323507200)))]; + tensor decoder_layers_4_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(323511360))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(324560000))), name = tensor("decoder_layers_4_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_4_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_4_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(324561088)))]; + tensor decoder_layers_4_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(324565248))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(325613888))), name = tensor("decoder_layers_4_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_4_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_4_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(325614976)))]; + tensor decoder_layers_4_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(325619136))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(326667776))), name = tensor("decoder_layers_4_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_4_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_4_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(326668864)))]; + tensor decoder_layers_4_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_4_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(326673024)))]; + tensor decoder_layers_4_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_4_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(326677184)))]; + tensor decoder_layers_4_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(326681344))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(327729984))), name = tensor("decoder_layers_4_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_4_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_4_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(327731072)))]; + tensor decoder_layers_4_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(327735232))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(328783872))), name = tensor("decoder_layers_4_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_4_final_layer_norm_bias = const()[name = tensor("decoder_layers_4_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(328784960)))]; + tensor decoder_layers_4_final_layer_norm_weight = const()[name = tensor("decoder_layers_4_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(328789120)))]; + tensor decoder_layers_4_fc1_bias = const()[name = tensor("decoder_layers_4_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(328793280)))]; + tensor decoder_layers_4_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(328809728))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(333004096))), name = tensor("decoder_layers_4_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_4_fc2_bias = const()[name = tensor("decoder_layers_4_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(333005184)))]; + tensor decoder_layers_4_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(333009344))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(337203712))), name = tensor("decoder_layers_4_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_5_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_5_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(337204800)))]; + tensor decoder_layers_5_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_5_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(337208960)))]; + tensor decoder_layers_5_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_5_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(337213120)))]; + tensor decoder_layers_5_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(337217280))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(338265920))), name = tensor("decoder_layers_5_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_5_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_5_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(338267008)))]; + tensor decoder_layers_5_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(338271168))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(339319808))), name = tensor("decoder_layers_5_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_5_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_5_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(339320896)))]; + tensor decoder_layers_5_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(339325056))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(340373696))), name = tensor("decoder_layers_5_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_5_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_5_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(340374784)))]; + tensor decoder_layers_5_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(340378944))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(341427584))), name = tensor("decoder_layers_5_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_5_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_5_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(341428672)))]; + tensor decoder_layers_5_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_5_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(341432832)))]; + tensor decoder_layers_5_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_5_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(341436992)))]; + tensor decoder_layers_5_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(341441152))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(342489792))), name = tensor("decoder_layers_5_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_5_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_5_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(342490880)))]; + tensor decoder_layers_5_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(342495040))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(343543680))), name = tensor("decoder_layers_5_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_5_final_layer_norm_bias = const()[name = tensor("decoder_layers_5_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(343544768)))]; + tensor decoder_layers_5_final_layer_norm_weight = const()[name = tensor("decoder_layers_5_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(343548928)))]; + tensor decoder_layers_5_fc1_bias = const()[name = tensor("decoder_layers_5_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(343553088)))]; + tensor decoder_layers_5_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(343569536))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(347763904))), name = tensor("decoder_layers_5_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_5_fc2_bias = const()[name = tensor("decoder_layers_5_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(347764992)))]; + tensor decoder_layers_5_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(347769152))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(351963520))), name = tensor("decoder_layers_5_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_6_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_6_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(351964608)))]; + tensor decoder_layers_6_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_6_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(351968768)))]; + tensor decoder_layers_6_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_6_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(351972928)))]; + tensor decoder_layers_6_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(351977088))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(353025728))), name = tensor("decoder_layers_6_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_6_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_6_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(353026816)))]; + tensor decoder_layers_6_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(353030976))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(354079616))), name = tensor("decoder_layers_6_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_6_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_6_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(354080704)))]; + tensor decoder_layers_6_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(354084864))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(355133504))), name = tensor("decoder_layers_6_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_6_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_6_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(355134592)))]; + tensor decoder_layers_6_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(355138752))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(356187392))), name = tensor("decoder_layers_6_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_6_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_6_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(356188480)))]; + tensor decoder_layers_6_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_6_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(356192640)))]; + tensor decoder_layers_6_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_6_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(356196800)))]; + tensor decoder_layers_6_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(356200960))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(357249600))), name = tensor("decoder_layers_6_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_6_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_6_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(357250688)))]; + tensor decoder_layers_6_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(357254848))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(358303488))), name = tensor("decoder_layers_6_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_6_final_layer_norm_bias = const()[name = tensor("decoder_layers_6_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(358304576)))]; + tensor decoder_layers_6_final_layer_norm_weight = const()[name = tensor("decoder_layers_6_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(358308736)))]; + tensor decoder_layers_6_fc1_bias = const()[name = tensor("decoder_layers_6_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(358312896)))]; + tensor decoder_layers_6_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(358329344))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(362523712))), name = tensor("decoder_layers_6_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_6_fc2_bias = const()[name = tensor("decoder_layers_6_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(362524800)))]; + tensor decoder_layers_6_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(362528960))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(366723328))), name = tensor("decoder_layers_6_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_7_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_7_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(366724416)))]; + tensor decoder_layers_7_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_7_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(366728576)))]; + tensor decoder_layers_7_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_7_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(366732736)))]; + tensor decoder_layers_7_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(366736896))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(367785536))), name = tensor("decoder_layers_7_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_7_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_7_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(367786624)))]; + tensor decoder_layers_7_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(367790784))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(368839424))), name = tensor("decoder_layers_7_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_7_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_7_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(368840512)))]; + tensor decoder_layers_7_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(368844672))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(369893312))), name = tensor("decoder_layers_7_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_7_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_7_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(369894400)))]; + tensor decoder_layers_7_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(369898560))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(370947200))), name = tensor("decoder_layers_7_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_7_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_7_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(370948288)))]; + tensor decoder_layers_7_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_7_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(370952448)))]; + tensor decoder_layers_7_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_7_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(370956608)))]; + tensor decoder_layers_7_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(370960768))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(372009408))), name = tensor("decoder_layers_7_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_7_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_7_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(372010496)))]; + tensor decoder_layers_7_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(372014656))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(362523712))), name = tensor("decoder_layers_7_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_7_final_layer_norm_bias = const()[name = tensor("decoder_layers_7_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(373063296)))]; + tensor decoder_layers_7_final_layer_norm_weight = const()[name = tensor("decoder_layers_7_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(373067456)))]; + tensor decoder_layers_7_fc1_bias = const()[name = tensor("decoder_layers_7_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(373071616)))]; + tensor decoder_layers_7_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(373088064))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263406848))), name = tensor("decoder_layers_7_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_7_fc2_bias = const()[name = tensor("decoder_layers_7_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(377282432)))]; + tensor decoder_layers_7_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(377286592))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(381480960))), name = tensor("decoder_layers_7_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_8_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_8_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(381482048)))]; + tensor decoder_layers_8_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_8_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(381486208)))]; + tensor decoder_layers_8_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_8_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(381490368)))]; + tensor decoder_layers_8_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(381494528))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(382543168))), name = tensor("decoder_layers_8_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_8_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_8_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(382544256)))]; + tensor decoder_layers_8_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(382548416))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(383597056))), name = tensor("decoder_layers_8_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_8_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_8_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(383598144)))]; + tensor decoder_layers_8_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(383602304))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263406848))), name = tensor("decoder_layers_8_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_8_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_8_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(384650944)))]; + tensor decoder_layers_8_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(384655104))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(385703744))), name = tensor("decoder_layers_8_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_8_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_8_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(385704832)))]; + tensor decoder_layers_8_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_8_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(385708992)))]; + tensor decoder_layers_8_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_8_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(385713152)))]; + tensor decoder_layers_8_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(385717312))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(386765952))), name = tensor("decoder_layers_8_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_8_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_8_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(386767040)))]; + tensor decoder_layers_8_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(386771200))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(387819840))), name = tensor("decoder_layers_8_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_8_final_layer_norm_bias = const()[name = tensor("decoder_layers_8_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(387820928)))]; + tensor decoder_layers_8_final_layer_norm_weight = const()[name = tensor("decoder_layers_8_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(387825088)))]; + tensor decoder_layers_8_fc1_bias = const()[name = tensor("decoder_layers_8_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(387829248)))]; + tensor decoder_layers_8_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(387845696))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(392040064))), name = tensor("decoder_layers_8_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_8_fc2_bias = const()[name = tensor("decoder_layers_8_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(392041152)))]; + tensor decoder_layers_8_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(392045312))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(396239680))), name = tensor("decoder_layers_8_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_9_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_9_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(396240768)))]; + tensor decoder_layers_9_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_9_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(396244928)))]; + tensor decoder_layers_9_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_9_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(396249088)))]; + tensor decoder_layers_9_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(396253248))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(397301888))), name = tensor("decoder_layers_9_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_9_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_9_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(397302976)))]; + tensor decoder_layers_9_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(397307136))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263406848))), name = tensor("decoder_layers_9_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_9_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_9_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(398355776)))]; + tensor decoder_layers_9_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(398359936))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(399408576))), name = tensor("decoder_layers_9_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_9_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_9_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(399409664)))]; + tensor decoder_layers_9_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(399413824))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(400462464))), name = tensor("decoder_layers_9_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_9_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_9_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(400463552)))]; + tensor decoder_layers_9_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_9_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(400467712)))]; + tensor decoder_layers_9_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_9_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(400471872)))]; + tensor decoder_layers_9_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(400476032))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(401524672))), name = tensor("decoder_layers_9_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_9_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_9_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(401525760)))]; + tensor decoder_layers_9_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(401529920))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(402578560))), name = tensor("decoder_layers_9_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_9_final_layer_norm_bias = const()[name = tensor("decoder_layers_9_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(402579648)))]; + tensor decoder_layers_9_final_layer_norm_weight = const()[name = tensor("decoder_layers_9_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(402583808)))]; + tensor decoder_layers_9_fc1_bias = const()[name = tensor("decoder_layers_9_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(402587968)))]; + tensor decoder_layers_9_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(402604416))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(406798784))), name = tensor("decoder_layers_9_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_9_fc2_bias = const()[name = tensor("decoder_layers_9_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(406799872)))]; + tensor decoder_layers_9_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(406804032))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(410998400))), name = tensor("decoder_layers_9_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_10_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_10_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(410999488)))]; + tensor decoder_layers_10_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_10_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(411003648)))]; + tensor decoder_layers_10_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_10_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(411007808)))]; + tensor decoder_layers_10_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(411011968))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(412060608))), name = tensor("decoder_layers_10_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_10_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_10_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(412061696)))]; + tensor decoder_layers_10_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(412065856))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(413114496))), name = tensor("decoder_layers_10_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_10_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_10_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(413115584)))]; + tensor decoder_layers_10_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(413119744))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(414168384))), name = tensor("decoder_layers_10_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_10_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_10_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(414169472)))]; + tensor decoder_layers_10_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(414173632))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(415222272))), name = tensor("decoder_layers_10_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_10_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_10_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(415223360)))]; + tensor decoder_layers_10_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_10_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(415227520)))]; + tensor decoder_layers_10_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_10_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(415231680)))]; + tensor decoder_layers_10_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(415235840))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(416284480))), name = tensor("decoder_layers_10_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_10_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_10_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(416285568)))]; + tensor decoder_layers_10_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(416289728))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(417338368))), name = tensor("decoder_layers_10_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_10_final_layer_norm_bias = const()[name = tensor("decoder_layers_10_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(417339456)))]; + tensor decoder_layers_10_final_layer_norm_weight = const()[name = tensor("decoder_layers_10_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(417343616)))]; + tensor decoder_layers_10_fc1_bias = const()[name = tensor("decoder_layers_10_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(417347776)))]; + tensor decoder_layers_10_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(417364224))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(421558592))), name = tensor("decoder_layers_10_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_10_fc2_bias = const()[name = tensor("decoder_layers_10_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(421559680)))]; + tensor decoder_layers_10_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(421563840))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(425758208))), name = tensor("decoder_layers_10_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_11_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_11_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(425759296)))]; + tensor decoder_layers_11_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_11_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(425763456)))]; + tensor decoder_layers_11_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_11_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(425767616)))]; + tensor decoder_layers_11_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(425771776))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(426820416))), name = tensor("decoder_layers_11_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_11_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_11_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(426821504)))]; + tensor decoder_layers_11_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(426825664))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(427874304))), name = tensor("decoder_layers_11_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_11_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_11_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(427875392)))]; + tensor decoder_layers_11_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(427879552))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(428928192))), name = tensor("decoder_layers_11_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_11_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_11_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(428929280)))]; + tensor decoder_layers_11_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(428933440))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(429982080))), name = tensor("decoder_layers_11_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_11_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_11_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(429983168)))]; + tensor decoder_layers_11_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_11_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(429987328)))]; + tensor decoder_layers_11_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_11_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(429991488)))]; + tensor decoder_layers_11_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(429995648))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(431044288))), name = tensor("decoder_layers_11_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_11_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_11_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(431045376)))]; + tensor decoder_layers_11_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(431049536))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(432098176))), name = tensor("decoder_layers_11_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_11_final_layer_norm_bias = const()[name = tensor("decoder_layers_11_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(432099264)))]; + tensor decoder_layers_11_final_layer_norm_weight = const()[name = tensor("decoder_layers_11_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(432103424)))]; + tensor decoder_layers_11_fc1_bias = const()[name = tensor("decoder_layers_11_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(432107584)))]; + tensor decoder_layers_11_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(432124032))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(436318400))), name = tensor("decoder_layers_11_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_11_fc2_bias = const()[name = tensor("decoder_layers_11_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(436319488)))]; + tensor decoder_layers_11_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(436323648))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(282390528))), name = tensor("decoder_layers_11_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layer_norm_bias = const()[name = tensor("decoder_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(440518016)))]; + tensor decoder_layer_norm_weight = const()[name = tensor("decoder_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(440522176)))]; + tensor key_states_1_begin_0 = const()[name = tensor("key_states_1_begin_0"), val = tensor([0, 0, 0, 0, 0])]; + tensor key_states_1_end_0 = const()[name = tensor("key_states_1_end_0"), val = tensor([1, 1, 16, 0, 64])]; + tensor key_states_1_end_mask_0 = const()[name = tensor("key_states_1_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_1_squeeze_mask_0 = const()[name = tensor("key_states_1_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor cast_0 = cast(dtype = cast_0_dtype_0, x = past_self_key)[name = tensor("cast_7")]; + tensor key_states_1 = slice_by_index(begin = key_states_1_begin_0, end = key_states_1_end_0, end_mask = key_states_1_end_mask_0, squeeze_mask = key_states_1_squeeze_mask_0, x = cast_0)[name = tensor("key_states_1")]; + tensor value_states_1_begin_0 = const()[name = tensor("value_states_1_begin_0"), val = tensor([0, 0, 0, 0, 0])]; + tensor value_states_1_end_0 = const()[name = tensor("value_states_1_end_0"), val = tensor([1, 1, 16, 0, 64])]; + tensor value_states_1_end_mask_0 = const()[name = tensor("value_states_1_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_1_squeeze_mask_0 = const()[name = tensor("value_states_1_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor cast_1 = cast(dtype = cast_1_dtype_0, x = past_self_value)[name = tensor("cast_6")]; + tensor value_states_1 = slice_by_index(begin = value_states_1_begin_0, end = value_states_1_end_0, end_mask = value_states_1_end_mask_0, squeeze_mask = value_states_1_squeeze_mask_0, x = cast_1)[name = tensor("value_states_1")]; + tensor key_states_3_begin_0 = const()[name = tensor("key_states_3_begin_0"), val = tensor([0, 0, 0, 0, 0])]; + tensor key_states_3_end_0 = const()[name = tensor("key_states_3_end_0"), val = tensor([1, 1, 16, 0, 64])]; + tensor key_states_3_end_mask_0 = const()[name = tensor("key_states_3_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_3_squeeze_mask_0 = const()[name = tensor("key_states_3_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor cast_2 = cast(dtype = cast_2_dtype_0, x = past_cross_key)[name = tensor("cast_5")]; + tensor key_states_3 = slice_by_index(begin = key_states_3_begin_0, end = key_states_3_end_0, end_mask = key_states_3_end_mask_0, squeeze_mask = key_states_3_squeeze_mask_0, x = cast_2)[name = tensor("key_states_3")]; + tensor value_states_3_begin_0 = const()[name = tensor("value_states_3_begin_0"), val = tensor([0, 0, 0, 0, 0])]; + tensor value_states_3_end_0 = const()[name = tensor("value_states_3_end_0"), val = tensor([1, 1, 16, 0, 64])]; + tensor value_states_3_end_mask_0 = const()[name = tensor("value_states_3_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_3_squeeze_mask_0 = const()[name = tensor("value_states_3_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor cast_3 = cast(dtype = cast_3_dtype_0, x = past_cross_value)[name = tensor("cast_4")]; + tensor value_states_3 = slice_by_index(begin = value_states_3_begin_0, end = value_states_3_end_0, end_mask = value_states_3_end_mask_0, squeeze_mask = value_states_3_squeeze_mask_0, x = cast_3)[name = tensor("value_states_3")]; + tensor key_states_5_begin_0 = const()[name = tensor("key_states_5_begin_0"), val = tensor([1, 0, 0, 0, 0])]; + tensor key_states_5_end_0 = const()[name = tensor("key_states_5_end_0"), val = tensor([2, 1, 16, 0, 64])]; + tensor key_states_5_end_mask_0 = const()[name = tensor("key_states_5_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_5_squeeze_mask_0 = const()[name = tensor("key_states_5_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_5 = slice_by_index(begin = key_states_5_begin_0, end = key_states_5_end_0, end_mask = key_states_5_end_mask_0, squeeze_mask = key_states_5_squeeze_mask_0, x = cast_0)[name = tensor("key_states_5")]; + tensor value_states_5_begin_0 = const()[name = tensor("value_states_5_begin_0"), val = tensor([1, 0, 0, 0, 0])]; + tensor value_states_5_end_0 = const()[name = tensor("value_states_5_end_0"), val = tensor([2, 1, 16, 0, 64])]; + tensor value_states_5_end_mask_0 = const()[name = tensor("value_states_5_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_5_squeeze_mask_0 = const()[name = tensor("value_states_5_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_5 = slice_by_index(begin = value_states_5_begin_0, end = value_states_5_end_0, end_mask = value_states_5_end_mask_0, squeeze_mask = value_states_5_squeeze_mask_0, x = cast_1)[name = tensor("value_states_5")]; + tensor key_states_7_begin_0 = const()[name = tensor("key_states_7_begin_0"), val = tensor([1, 0, 0, 0, 0])]; + tensor key_states_7_end_0 = const()[name = tensor("key_states_7_end_0"), val = tensor([2, 1, 16, 0, 64])]; + tensor key_states_7_end_mask_0 = const()[name = tensor("key_states_7_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_7_squeeze_mask_0 = const()[name = tensor("key_states_7_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_7 = slice_by_index(begin = key_states_7_begin_0, end = key_states_7_end_0, end_mask = key_states_7_end_mask_0, squeeze_mask = key_states_7_squeeze_mask_0, x = cast_2)[name = tensor("key_states_7")]; + tensor value_states_7_begin_0 = const()[name = tensor("value_states_7_begin_0"), val = tensor([1, 0, 0, 0, 0])]; + tensor value_states_7_end_0 = const()[name = tensor("value_states_7_end_0"), val = tensor([2, 1, 16, 0, 64])]; + tensor value_states_7_end_mask_0 = const()[name = tensor("value_states_7_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_7_squeeze_mask_0 = const()[name = tensor("value_states_7_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_7 = slice_by_index(begin = value_states_7_begin_0, end = value_states_7_end_0, end_mask = value_states_7_end_mask_0, squeeze_mask = value_states_7_squeeze_mask_0, x = cast_3)[name = tensor("value_states_7")]; + tensor key_states_9_begin_0 = const()[name = tensor("key_states_9_begin_0"), val = tensor([2, 0, 0, 0, 0])]; + tensor key_states_9_end_0 = const()[name = tensor("key_states_9_end_0"), val = tensor([3, 1, 16, 0, 64])]; + tensor key_states_9_end_mask_0 = const()[name = tensor("key_states_9_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_9_squeeze_mask_0 = const()[name = tensor("key_states_9_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_9 = slice_by_index(begin = key_states_9_begin_0, end = key_states_9_end_0, end_mask = key_states_9_end_mask_0, squeeze_mask = key_states_9_squeeze_mask_0, x = cast_0)[name = tensor("key_states_9")]; + tensor value_states_9_begin_0 = const()[name = tensor("value_states_9_begin_0"), val = tensor([2, 0, 0, 0, 0])]; + tensor value_states_9_end_0 = const()[name = tensor("value_states_9_end_0"), val = tensor([3, 1, 16, 0, 64])]; + tensor value_states_9_end_mask_0 = const()[name = tensor("value_states_9_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_9_squeeze_mask_0 = const()[name = tensor("value_states_9_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_9 = slice_by_index(begin = value_states_9_begin_0, end = value_states_9_end_0, end_mask = value_states_9_end_mask_0, squeeze_mask = value_states_9_squeeze_mask_0, x = cast_1)[name = tensor("value_states_9")]; + tensor key_states_11_begin_0 = const()[name = tensor("key_states_11_begin_0"), val = tensor([2, 0, 0, 0, 0])]; + tensor key_states_11_end_0 = const()[name = tensor("key_states_11_end_0"), val = tensor([3, 1, 16, 0, 64])]; + tensor key_states_11_end_mask_0 = const()[name = tensor("key_states_11_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_11_squeeze_mask_0 = const()[name = tensor("key_states_11_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_11 = slice_by_index(begin = key_states_11_begin_0, end = key_states_11_end_0, end_mask = key_states_11_end_mask_0, squeeze_mask = key_states_11_squeeze_mask_0, x = cast_2)[name = tensor("key_states_11")]; + tensor value_states_11_begin_0 = const()[name = tensor("value_states_11_begin_0"), val = tensor([2, 0, 0, 0, 0])]; + tensor value_states_11_end_0 = const()[name = tensor("value_states_11_end_0"), val = tensor([3, 1, 16, 0, 64])]; + tensor value_states_11_end_mask_0 = const()[name = tensor("value_states_11_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_11_squeeze_mask_0 = const()[name = tensor("value_states_11_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_11 = slice_by_index(begin = value_states_11_begin_0, end = value_states_11_end_0, end_mask = value_states_11_end_mask_0, squeeze_mask = value_states_11_squeeze_mask_0, x = cast_3)[name = tensor("value_states_11")]; + tensor key_states_13_begin_0 = const()[name = tensor("key_states_13_begin_0"), val = tensor([3, 0, 0, 0, 0])]; + tensor key_states_13_end_0 = const()[name = tensor("key_states_13_end_0"), val = tensor([4, 1, 16, 0, 64])]; + tensor key_states_13_end_mask_0 = const()[name = tensor("key_states_13_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_13_squeeze_mask_0 = const()[name = tensor("key_states_13_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_13 = slice_by_index(begin = key_states_13_begin_0, end = key_states_13_end_0, end_mask = key_states_13_end_mask_0, squeeze_mask = key_states_13_squeeze_mask_0, x = cast_0)[name = tensor("key_states_13")]; + tensor value_states_13_begin_0 = const()[name = tensor("value_states_13_begin_0"), val = tensor([3, 0, 0, 0, 0])]; + tensor value_states_13_end_0 = const()[name = tensor("value_states_13_end_0"), val = tensor([4, 1, 16, 0, 64])]; + tensor value_states_13_end_mask_0 = const()[name = tensor("value_states_13_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_13_squeeze_mask_0 = const()[name = tensor("value_states_13_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_13 = slice_by_index(begin = value_states_13_begin_0, end = value_states_13_end_0, end_mask = value_states_13_end_mask_0, squeeze_mask = value_states_13_squeeze_mask_0, x = cast_1)[name = tensor("value_states_13")]; + tensor key_states_15_begin_0 = const()[name = tensor("key_states_15_begin_0"), val = tensor([3, 0, 0, 0, 0])]; + tensor key_states_15_end_0 = const()[name = tensor("key_states_15_end_0"), val = tensor([4, 1, 16, 0, 64])]; + tensor key_states_15_end_mask_0 = const()[name = tensor("key_states_15_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_15_squeeze_mask_0 = const()[name = tensor("key_states_15_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_15 = slice_by_index(begin = key_states_15_begin_0, end = key_states_15_end_0, end_mask = key_states_15_end_mask_0, squeeze_mask = key_states_15_squeeze_mask_0, x = cast_2)[name = tensor("key_states_15")]; + tensor value_states_15_begin_0 = const()[name = tensor("value_states_15_begin_0"), val = tensor([3, 0, 0, 0, 0])]; + tensor value_states_15_end_0 = const()[name = tensor("value_states_15_end_0"), val = tensor([4, 1, 16, 0, 64])]; + tensor value_states_15_end_mask_0 = const()[name = tensor("value_states_15_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_15_squeeze_mask_0 = const()[name = tensor("value_states_15_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_15 = slice_by_index(begin = value_states_15_begin_0, end = value_states_15_end_0, end_mask = value_states_15_end_mask_0, squeeze_mask = value_states_15_squeeze_mask_0, x = cast_3)[name = tensor("value_states_15")]; + tensor key_states_17_begin_0 = const()[name = tensor("key_states_17_begin_0"), val = tensor([4, 0, 0, 0, 0])]; + tensor key_states_17_end_0 = const()[name = tensor("key_states_17_end_0"), val = tensor([5, 1, 16, 0, 64])]; + tensor key_states_17_end_mask_0 = const()[name = tensor("key_states_17_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_17_squeeze_mask_0 = const()[name = tensor("key_states_17_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_17 = slice_by_index(begin = key_states_17_begin_0, end = key_states_17_end_0, end_mask = key_states_17_end_mask_0, squeeze_mask = key_states_17_squeeze_mask_0, x = cast_0)[name = tensor("key_states_17")]; + tensor value_states_17_begin_0 = const()[name = tensor("value_states_17_begin_0"), val = tensor([4, 0, 0, 0, 0])]; + tensor value_states_17_end_0 = const()[name = tensor("value_states_17_end_0"), val = tensor([5, 1, 16, 0, 64])]; + tensor value_states_17_end_mask_0 = const()[name = tensor("value_states_17_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_17_squeeze_mask_0 = const()[name = tensor("value_states_17_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_17 = slice_by_index(begin = value_states_17_begin_0, end = value_states_17_end_0, end_mask = value_states_17_end_mask_0, squeeze_mask = value_states_17_squeeze_mask_0, x = cast_1)[name = tensor("value_states_17")]; + tensor key_states_19_begin_0 = const()[name = tensor("key_states_19_begin_0"), val = tensor([4, 0, 0, 0, 0])]; + tensor key_states_19_end_0 = const()[name = tensor("key_states_19_end_0"), val = tensor([5, 1, 16, 0, 64])]; + tensor key_states_19_end_mask_0 = const()[name = tensor("key_states_19_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_19_squeeze_mask_0 = const()[name = tensor("key_states_19_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_19 = slice_by_index(begin = key_states_19_begin_0, end = key_states_19_end_0, end_mask = key_states_19_end_mask_0, squeeze_mask = key_states_19_squeeze_mask_0, x = cast_2)[name = tensor("key_states_19")]; + tensor value_states_19_begin_0 = const()[name = tensor("value_states_19_begin_0"), val = tensor([4, 0, 0, 0, 0])]; + tensor value_states_19_end_0 = const()[name = tensor("value_states_19_end_0"), val = tensor([5, 1, 16, 0, 64])]; + tensor value_states_19_end_mask_0 = const()[name = tensor("value_states_19_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_19_squeeze_mask_0 = const()[name = tensor("value_states_19_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_19 = slice_by_index(begin = value_states_19_begin_0, end = value_states_19_end_0, end_mask = value_states_19_end_mask_0, squeeze_mask = value_states_19_squeeze_mask_0, x = cast_3)[name = tensor("value_states_19")]; + tensor key_states_21_begin_0 = const()[name = tensor("key_states_21_begin_0"), val = tensor([5, 0, 0, 0, 0])]; + tensor key_states_21_end_0 = const()[name = tensor("key_states_21_end_0"), val = tensor([6, 1, 16, 0, 64])]; + tensor key_states_21_end_mask_0 = const()[name = tensor("key_states_21_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_21_squeeze_mask_0 = const()[name = tensor("key_states_21_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_21 = slice_by_index(begin = key_states_21_begin_0, end = key_states_21_end_0, end_mask = key_states_21_end_mask_0, squeeze_mask = key_states_21_squeeze_mask_0, x = cast_0)[name = tensor("key_states_21")]; + tensor value_states_21_begin_0 = const()[name = tensor("value_states_21_begin_0"), val = tensor([5, 0, 0, 0, 0])]; + tensor value_states_21_end_0 = const()[name = tensor("value_states_21_end_0"), val = tensor([6, 1, 16, 0, 64])]; + tensor value_states_21_end_mask_0 = const()[name = tensor("value_states_21_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_21_squeeze_mask_0 = const()[name = tensor("value_states_21_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_21 = slice_by_index(begin = value_states_21_begin_0, end = value_states_21_end_0, end_mask = value_states_21_end_mask_0, squeeze_mask = value_states_21_squeeze_mask_0, x = cast_1)[name = tensor("value_states_21")]; + tensor key_states_23_begin_0 = const()[name = tensor("key_states_23_begin_0"), val = tensor([5, 0, 0, 0, 0])]; + tensor key_states_23_end_0 = const()[name = tensor("key_states_23_end_0"), val = tensor([6, 1, 16, 0, 64])]; + tensor key_states_23_end_mask_0 = const()[name = tensor("key_states_23_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_23_squeeze_mask_0 = const()[name = tensor("key_states_23_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_23 = slice_by_index(begin = key_states_23_begin_0, end = key_states_23_end_0, end_mask = key_states_23_end_mask_0, squeeze_mask = key_states_23_squeeze_mask_0, x = cast_2)[name = tensor("key_states_23")]; + tensor value_states_23_begin_0 = const()[name = tensor("value_states_23_begin_0"), val = tensor([5, 0, 0, 0, 0])]; + tensor value_states_23_end_0 = const()[name = tensor("value_states_23_end_0"), val = tensor([6, 1, 16, 0, 64])]; + tensor value_states_23_end_mask_0 = const()[name = tensor("value_states_23_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_23_squeeze_mask_0 = const()[name = tensor("value_states_23_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_23 = slice_by_index(begin = value_states_23_begin_0, end = value_states_23_end_0, end_mask = value_states_23_end_mask_0, squeeze_mask = value_states_23_squeeze_mask_0, x = cast_3)[name = tensor("value_states_23")]; + tensor key_states_25_begin_0 = const()[name = tensor("key_states_25_begin_0"), val = tensor([6, 0, 0, 0, 0])]; + tensor key_states_25_end_0 = const()[name = tensor("key_states_25_end_0"), val = tensor([7, 1, 16, 0, 64])]; + tensor key_states_25_end_mask_0 = const()[name = tensor("key_states_25_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_25_squeeze_mask_0 = const()[name = tensor("key_states_25_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_25 = slice_by_index(begin = key_states_25_begin_0, end = key_states_25_end_0, end_mask = key_states_25_end_mask_0, squeeze_mask = key_states_25_squeeze_mask_0, x = cast_0)[name = tensor("key_states_25")]; + tensor value_states_25_begin_0 = const()[name = tensor("value_states_25_begin_0"), val = tensor([6, 0, 0, 0, 0])]; + tensor value_states_25_end_0 = const()[name = tensor("value_states_25_end_0"), val = tensor([7, 1, 16, 0, 64])]; + tensor value_states_25_end_mask_0 = const()[name = tensor("value_states_25_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_25_squeeze_mask_0 = const()[name = tensor("value_states_25_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_25 = slice_by_index(begin = value_states_25_begin_0, end = value_states_25_end_0, end_mask = value_states_25_end_mask_0, squeeze_mask = value_states_25_squeeze_mask_0, x = cast_1)[name = tensor("value_states_25")]; + tensor key_states_27_begin_0 = const()[name = tensor("key_states_27_begin_0"), val = tensor([6, 0, 0, 0, 0])]; + tensor key_states_27_end_0 = const()[name = tensor("key_states_27_end_0"), val = tensor([7, 1, 16, 0, 64])]; + tensor key_states_27_end_mask_0 = const()[name = tensor("key_states_27_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_27_squeeze_mask_0 = const()[name = tensor("key_states_27_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_27 = slice_by_index(begin = key_states_27_begin_0, end = key_states_27_end_0, end_mask = key_states_27_end_mask_0, squeeze_mask = key_states_27_squeeze_mask_0, x = cast_2)[name = tensor("key_states_27")]; + tensor value_states_27_begin_0 = const()[name = tensor("value_states_27_begin_0"), val = tensor([6, 0, 0, 0, 0])]; + tensor value_states_27_end_0 = const()[name = tensor("value_states_27_end_0"), val = tensor([7, 1, 16, 0, 64])]; + tensor value_states_27_end_mask_0 = const()[name = tensor("value_states_27_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_27_squeeze_mask_0 = const()[name = tensor("value_states_27_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_27 = slice_by_index(begin = value_states_27_begin_0, end = value_states_27_end_0, end_mask = value_states_27_end_mask_0, squeeze_mask = value_states_27_squeeze_mask_0, x = cast_3)[name = tensor("value_states_27")]; + tensor key_states_29_begin_0 = const()[name = tensor("key_states_29_begin_0"), val = tensor([7, 0, 0, 0, 0])]; + tensor key_states_29_end_0 = const()[name = tensor("key_states_29_end_0"), val = tensor([8, 1, 16, 0, 64])]; + tensor key_states_29_end_mask_0 = const()[name = tensor("key_states_29_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_29_squeeze_mask_0 = const()[name = tensor("key_states_29_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_29 = slice_by_index(begin = key_states_29_begin_0, end = key_states_29_end_0, end_mask = key_states_29_end_mask_0, squeeze_mask = key_states_29_squeeze_mask_0, x = cast_0)[name = tensor("key_states_29")]; + tensor value_states_29_begin_0 = const()[name = tensor("value_states_29_begin_0"), val = tensor([7, 0, 0, 0, 0])]; + tensor value_states_29_end_0 = const()[name = tensor("value_states_29_end_0"), val = tensor([8, 1, 16, 0, 64])]; + tensor value_states_29_end_mask_0 = const()[name = tensor("value_states_29_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_29_squeeze_mask_0 = const()[name = tensor("value_states_29_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_29 = slice_by_index(begin = value_states_29_begin_0, end = value_states_29_end_0, end_mask = value_states_29_end_mask_0, squeeze_mask = value_states_29_squeeze_mask_0, x = cast_1)[name = tensor("value_states_29")]; + tensor key_states_31_begin_0 = const()[name = tensor("key_states_31_begin_0"), val = tensor([7, 0, 0, 0, 0])]; + tensor key_states_31_end_0 = const()[name = tensor("key_states_31_end_0"), val = tensor([8, 1, 16, 0, 64])]; + tensor key_states_31_end_mask_0 = const()[name = tensor("key_states_31_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_31_squeeze_mask_0 = const()[name = tensor("key_states_31_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_31 = slice_by_index(begin = key_states_31_begin_0, end = key_states_31_end_0, end_mask = key_states_31_end_mask_0, squeeze_mask = key_states_31_squeeze_mask_0, x = cast_2)[name = tensor("key_states_31")]; + tensor value_states_31_begin_0 = const()[name = tensor("value_states_31_begin_0"), val = tensor([7, 0, 0, 0, 0])]; + tensor value_states_31_end_0 = const()[name = tensor("value_states_31_end_0"), val = tensor([8, 1, 16, 0, 64])]; + tensor value_states_31_end_mask_0 = const()[name = tensor("value_states_31_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_31_squeeze_mask_0 = const()[name = tensor("value_states_31_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_31 = slice_by_index(begin = value_states_31_begin_0, end = value_states_31_end_0, end_mask = value_states_31_end_mask_0, squeeze_mask = value_states_31_squeeze_mask_0, x = cast_3)[name = tensor("value_states_31")]; + tensor key_states_33_begin_0 = const()[name = tensor("key_states_33_begin_0"), val = tensor([8, 0, 0, 0, 0])]; + tensor key_states_33_end_0 = const()[name = tensor("key_states_33_end_0"), val = tensor([9, 1, 16, 0, 64])]; + tensor key_states_33_end_mask_0 = const()[name = tensor("key_states_33_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_33_squeeze_mask_0 = const()[name = tensor("key_states_33_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_33 = slice_by_index(begin = key_states_33_begin_0, end = key_states_33_end_0, end_mask = key_states_33_end_mask_0, squeeze_mask = key_states_33_squeeze_mask_0, x = cast_0)[name = tensor("key_states_33")]; + tensor value_states_33_begin_0 = const()[name = tensor("value_states_33_begin_0"), val = tensor([8, 0, 0, 0, 0])]; + tensor value_states_33_end_0 = const()[name = tensor("value_states_33_end_0"), val = tensor([9, 1, 16, 0, 64])]; + tensor value_states_33_end_mask_0 = const()[name = tensor("value_states_33_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_33_squeeze_mask_0 = const()[name = tensor("value_states_33_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_33 = slice_by_index(begin = value_states_33_begin_0, end = value_states_33_end_0, end_mask = value_states_33_end_mask_0, squeeze_mask = value_states_33_squeeze_mask_0, x = cast_1)[name = tensor("value_states_33")]; + tensor key_states_35_begin_0 = const()[name = tensor("key_states_35_begin_0"), val = tensor([8, 0, 0, 0, 0])]; + tensor key_states_35_end_0 = const()[name = tensor("key_states_35_end_0"), val = tensor([9, 1, 16, 0, 64])]; + tensor key_states_35_end_mask_0 = const()[name = tensor("key_states_35_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_35_squeeze_mask_0 = const()[name = tensor("key_states_35_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_35 = slice_by_index(begin = key_states_35_begin_0, end = key_states_35_end_0, end_mask = key_states_35_end_mask_0, squeeze_mask = key_states_35_squeeze_mask_0, x = cast_2)[name = tensor("key_states_35")]; + tensor value_states_35_begin_0 = const()[name = tensor("value_states_35_begin_0"), val = tensor([8, 0, 0, 0, 0])]; + tensor value_states_35_end_0 = const()[name = tensor("value_states_35_end_0"), val = tensor([9, 1, 16, 0, 64])]; + tensor value_states_35_end_mask_0 = const()[name = tensor("value_states_35_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_35_squeeze_mask_0 = const()[name = tensor("value_states_35_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_35 = slice_by_index(begin = value_states_35_begin_0, end = value_states_35_end_0, end_mask = value_states_35_end_mask_0, squeeze_mask = value_states_35_squeeze_mask_0, x = cast_3)[name = tensor("value_states_35")]; + tensor key_states_37_begin_0 = const()[name = tensor("key_states_37_begin_0"), val = tensor([9, 0, 0, 0, 0])]; + tensor key_states_37_end_0 = const()[name = tensor("key_states_37_end_0"), val = tensor([10, 1, 16, 0, 64])]; + tensor key_states_37_end_mask_0 = const()[name = tensor("key_states_37_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_37_squeeze_mask_0 = const()[name = tensor("key_states_37_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_37 = slice_by_index(begin = key_states_37_begin_0, end = key_states_37_end_0, end_mask = key_states_37_end_mask_0, squeeze_mask = key_states_37_squeeze_mask_0, x = cast_0)[name = tensor("key_states_37")]; + tensor value_states_37_begin_0 = const()[name = tensor("value_states_37_begin_0"), val = tensor([9, 0, 0, 0, 0])]; + tensor value_states_37_end_0 = const()[name = tensor("value_states_37_end_0"), val = tensor([10, 1, 16, 0, 64])]; + tensor value_states_37_end_mask_0 = const()[name = tensor("value_states_37_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_37_squeeze_mask_0 = const()[name = tensor("value_states_37_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_37 = slice_by_index(begin = value_states_37_begin_0, end = value_states_37_end_0, end_mask = value_states_37_end_mask_0, squeeze_mask = value_states_37_squeeze_mask_0, x = cast_1)[name = tensor("value_states_37")]; + tensor key_states_39_begin_0 = const()[name = tensor("key_states_39_begin_0"), val = tensor([9, 0, 0, 0, 0])]; + tensor key_states_39_end_0 = const()[name = tensor("key_states_39_end_0"), val = tensor([10, 1, 16, 0, 64])]; + tensor key_states_39_end_mask_0 = const()[name = tensor("key_states_39_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_39_squeeze_mask_0 = const()[name = tensor("key_states_39_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_39 = slice_by_index(begin = key_states_39_begin_0, end = key_states_39_end_0, end_mask = key_states_39_end_mask_0, squeeze_mask = key_states_39_squeeze_mask_0, x = cast_2)[name = tensor("key_states_39")]; + tensor value_states_39_begin_0 = const()[name = tensor("value_states_39_begin_0"), val = tensor([9, 0, 0, 0, 0])]; + tensor value_states_39_end_0 = const()[name = tensor("value_states_39_end_0"), val = tensor([10, 1, 16, 0, 64])]; + tensor value_states_39_end_mask_0 = const()[name = tensor("value_states_39_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_39_squeeze_mask_0 = const()[name = tensor("value_states_39_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_39 = slice_by_index(begin = value_states_39_begin_0, end = value_states_39_end_0, end_mask = value_states_39_end_mask_0, squeeze_mask = value_states_39_squeeze_mask_0, x = cast_3)[name = tensor("value_states_39")]; + tensor key_states_41_begin_0 = const()[name = tensor("key_states_41_begin_0"), val = tensor([10, 0, 0, 0, 0])]; + tensor key_states_41_end_0 = const()[name = tensor("key_states_41_end_0"), val = tensor([11, 1, 16, 0, 64])]; + tensor key_states_41_end_mask_0 = const()[name = tensor("key_states_41_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_41_squeeze_mask_0 = const()[name = tensor("key_states_41_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_41 = slice_by_index(begin = key_states_41_begin_0, end = key_states_41_end_0, end_mask = key_states_41_end_mask_0, squeeze_mask = key_states_41_squeeze_mask_0, x = cast_0)[name = tensor("key_states_41")]; + tensor value_states_41_begin_0 = const()[name = tensor("value_states_41_begin_0"), val = tensor([10, 0, 0, 0, 0])]; + tensor value_states_41_end_0 = const()[name = tensor("value_states_41_end_0"), val = tensor([11, 1, 16, 0, 64])]; + tensor value_states_41_end_mask_0 = const()[name = tensor("value_states_41_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_41_squeeze_mask_0 = const()[name = tensor("value_states_41_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_41 = slice_by_index(begin = value_states_41_begin_0, end = value_states_41_end_0, end_mask = value_states_41_end_mask_0, squeeze_mask = value_states_41_squeeze_mask_0, x = cast_1)[name = tensor("value_states_41")]; + tensor key_states_43_begin_0 = const()[name = tensor("key_states_43_begin_0"), val = tensor([10, 0, 0, 0, 0])]; + tensor key_states_43_end_0 = const()[name = tensor("key_states_43_end_0"), val = tensor([11, 1, 16, 0, 64])]; + tensor key_states_43_end_mask_0 = const()[name = tensor("key_states_43_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_43_squeeze_mask_0 = const()[name = tensor("key_states_43_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_43 = slice_by_index(begin = key_states_43_begin_0, end = key_states_43_end_0, end_mask = key_states_43_end_mask_0, squeeze_mask = key_states_43_squeeze_mask_0, x = cast_2)[name = tensor("key_states_43")]; + tensor value_states_43_begin_0 = const()[name = tensor("value_states_43_begin_0"), val = tensor([10, 0, 0, 0, 0])]; + tensor value_states_43_end_0 = const()[name = tensor("value_states_43_end_0"), val = tensor([11, 1, 16, 0, 64])]; + tensor value_states_43_end_mask_0 = const()[name = tensor("value_states_43_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_43_squeeze_mask_0 = const()[name = tensor("value_states_43_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_43 = slice_by_index(begin = value_states_43_begin_0, end = value_states_43_end_0, end_mask = value_states_43_end_mask_0, squeeze_mask = value_states_43_squeeze_mask_0, x = cast_3)[name = tensor("value_states_43")]; + tensor key_states_45_begin_0 = const()[name = tensor("key_states_45_begin_0"), val = tensor([11, 0, 0, 0, 0])]; + tensor key_states_45_end_0 = const()[name = tensor("key_states_45_end_0"), val = tensor([12, 1, 16, 0, 64])]; + tensor key_states_45_end_mask_0 = const()[name = tensor("key_states_45_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_45_squeeze_mask_0 = const()[name = tensor("key_states_45_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_45 = slice_by_index(begin = key_states_45_begin_0, end = key_states_45_end_0, end_mask = key_states_45_end_mask_0, squeeze_mask = key_states_45_squeeze_mask_0, x = cast_0)[name = tensor("key_states_45")]; + tensor value_states_45_begin_0 = const()[name = tensor("value_states_45_begin_0"), val = tensor([11, 0, 0, 0, 0])]; + tensor value_states_45_end_0 = const()[name = tensor("value_states_45_end_0"), val = tensor([12, 1, 16, 0, 64])]; + tensor value_states_45_end_mask_0 = const()[name = tensor("value_states_45_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_45_squeeze_mask_0 = const()[name = tensor("value_states_45_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_45 = slice_by_index(begin = value_states_45_begin_0, end = value_states_45_end_0, end_mask = value_states_45_end_mask_0, squeeze_mask = value_states_45_squeeze_mask_0, x = cast_1)[name = tensor("value_states_45")]; + tensor key_states_47_begin_0 = const()[name = tensor("key_states_47_begin_0"), val = tensor([11, 0, 0, 0, 0])]; + tensor key_states_47_end_0 = const()[name = tensor("key_states_47_end_0"), val = tensor([12, 1, 16, 0, 64])]; + tensor key_states_47_end_mask_0 = const()[name = tensor("key_states_47_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor key_states_47_squeeze_mask_0 = const()[name = tensor("key_states_47_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor key_states_47 = slice_by_index(begin = key_states_47_begin_0, end = key_states_47_end_0, end_mask = key_states_47_end_mask_0, squeeze_mask = key_states_47_squeeze_mask_0, x = cast_2)[name = tensor("key_states_47")]; + tensor value_states_47_begin_0 = const()[name = tensor("value_states_47_begin_0"), val = tensor([11, 0, 0, 0, 0])]; + tensor value_states_47_end_0 = const()[name = tensor("value_states_47_end_0"), val = tensor([12, 1, 16, 0, 64])]; + tensor value_states_47_end_mask_0 = const()[name = tensor("value_states_47_end_mask_0"), val = tensor([false, true, true, true, true])]; + tensor value_states_47_squeeze_mask_0 = const()[name = tensor("value_states_47_squeeze_mask_0"), val = tensor([true, false, false, false, false])]; + tensor value_states_47 = slice_by_index(begin = value_states_47_begin_0, end = value_states_47_end_0, end_mask = value_states_47_end_mask_0, squeeze_mask = value_states_47_squeeze_mask_0, x = cast_3)[name = tensor("value_states_47")]; + tensor var_173 = const()[name = tensor("op_173"), val = tensor(-2)]; + tensor var_174_interleave_0 = const()[name = tensor("op_174_interleave_0"), val = tensor(false)]; + tensor var_174 = concat(axis = var_173, interleave = var_174_interleave_0, values = key_states_1)[name = tensor("op_174")]; + tensor var_176 = const()[name = tensor("op_176"), val = tensor(-2)]; + tensor var_177_interleave_0 = const()[name = tensor("op_177_interleave_0"), val = tensor(false)]; + tensor var_177 = concat(axis = var_176, interleave = var_177_interleave_0, values = value_states_1)[name = tensor("op_177")]; + tensor var_195 = const()[name = tensor("op_195"), val = tensor(-2)]; + tensor key_3_interleave_0 = const()[name = tensor("key_3_interleave_0"), val = tensor(false)]; + tensor key_3 = concat(axis = var_195, interleave = key_3_interleave_0, values = key_states_3)[name = tensor("key_3")]; + tensor var_198 = const()[name = tensor("op_198"), val = tensor(-2)]; + tensor value_3_interleave_0 = const()[name = tensor("value_3_interleave_0"), val = tensor(false)]; + tensor value_3 = concat(axis = var_198, interleave = value_3_interleave_0, values = value_states_3)[name = tensor("value_3")]; + tensor var_217 = const()[name = tensor("op_217"), val = tensor(-2)]; + tensor var_218_interleave_0 = const()[name = tensor("op_218_interleave_0"), val = tensor(false)]; + tensor var_218 = concat(axis = var_217, interleave = var_218_interleave_0, values = key_states_5)[name = tensor("op_218")]; + tensor var_220 = const()[name = tensor("op_220"), val = tensor(-2)]; + tensor var_221_interleave_0 = const()[name = tensor("op_221_interleave_0"), val = tensor(false)]; + tensor var_221 = concat(axis = var_220, interleave = var_221_interleave_0, values = value_states_5)[name = tensor("op_221")]; + tensor var_239 = const()[name = tensor("op_239"), val = tensor(-2)]; + tensor key_7_interleave_0 = const()[name = tensor("key_7_interleave_0"), val = tensor(false)]; + tensor key_7 = concat(axis = var_239, interleave = key_7_interleave_0, values = key_states_7)[name = tensor("key_7")]; + tensor var_242 = const()[name = tensor("op_242"), val = tensor(-2)]; + tensor value_7_interleave_0 = const()[name = tensor("value_7_interleave_0"), val = tensor(false)]; + tensor value_7 = concat(axis = var_242, interleave = value_7_interleave_0, values = value_states_7)[name = tensor("value_7")]; + tensor var_261 = const()[name = tensor("op_261"), val = tensor(-2)]; + tensor var_262_interleave_0 = const()[name = tensor("op_262_interleave_0"), val = tensor(false)]; + tensor var_262 = concat(axis = var_261, interleave = var_262_interleave_0, values = key_states_9)[name = tensor("op_262")]; + tensor var_264 = const()[name = tensor("op_264"), val = tensor(-2)]; + tensor var_265_interleave_0 = const()[name = tensor("op_265_interleave_0"), val = tensor(false)]; + tensor var_265 = concat(axis = var_264, interleave = var_265_interleave_0, values = value_states_9)[name = tensor("op_265")]; + tensor var_283 = const()[name = tensor("op_283"), val = tensor(-2)]; + tensor key_11_interleave_0 = const()[name = tensor("key_11_interleave_0"), val = tensor(false)]; + tensor key_11 = concat(axis = var_283, interleave = key_11_interleave_0, values = key_states_11)[name = tensor("key_11")]; + tensor var_286 = const()[name = tensor("op_286"), val = tensor(-2)]; + tensor value_11_interleave_0 = const()[name = tensor("value_11_interleave_0"), val = tensor(false)]; + tensor value_11 = concat(axis = var_286, interleave = value_11_interleave_0, values = value_states_11)[name = tensor("value_11")]; + tensor var_305 = const()[name = tensor("op_305"), val = tensor(-2)]; + tensor var_306_interleave_0 = const()[name = tensor("op_306_interleave_0"), val = tensor(false)]; + tensor var_306 = concat(axis = var_305, interleave = var_306_interleave_0, values = key_states_13)[name = tensor("op_306")]; + tensor var_308 = const()[name = tensor("op_308"), val = tensor(-2)]; + tensor var_309_interleave_0 = const()[name = tensor("op_309_interleave_0"), val = tensor(false)]; + tensor var_309 = concat(axis = var_308, interleave = var_309_interleave_0, values = value_states_13)[name = tensor("op_309")]; + tensor var_327 = const()[name = tensor("op_327"), val = tensor(-2)]; + tensor key_15_interleave_0 = const()[name = tensor("key_15_interleave_0"), val = tensor(false)]; + tensor key_15 = concat(axis = var_327, interleave = key_15_interleave_0, values = key_states_15)[name = tensor("key_15")]; + tensor var_330 = const()[name = tensor("op_330"), val = tensor(-2)]; + tensor value_15_interleave_0 = const()[name = tensor("value_15_interleave_0"), val = tensor(false)]; + tensor value_15 = concat(axis = var_330, interleave = value_15_interleave_0, values = value_states_15)[name = tensor("value_15")]; + tensor var_349 = const()[name = tensor("op_349"), val = tensor(-2)]; + tensor var_350_interleave_0 = const()[name = tensor("op_350_interleave_0"), val = tensor(false)]; + tensor var_350 = concat(axis = var_349, interleave = var_350_interleave_0, values = key_states_17)[name = tensor("op_350")]; + tensor var_352 = const()[name = tensor("op_352"), val = tensor(-2)]; + tensor var_353_interleave_0 = const()[name = tensor("op_353_interleave_0"), val = tensor(false)]; + tensor var_353 = concat(axis = var_352, interleave = var_353_interleave_0, values = value_states_17)[name = tensor("op_353")]; + tensor var_371 = const()[name = tensor("op_371"), val = tensor(-2)]; + tensor key_19_interleave_0 = const()[name = tensor("key_19_interleave_0"), val = tensor(false)]; + tensor key_19 = concat(axis = var_371, interleave = key_19_interleave_0, values = key_states_19)[name = tensor("key_19")]; + tensor var_374 = const()[name = tensor("op_374"), val = tensor(-2)]; + tensor value_19_interleave_0 = const()[name = tensor("value_19_interleave_0"), val = tensor(false)]; + tensor value_19 = concat(axis = var_374, interleave = value_19_interleave_0, values = value_states_19)[name = tensor("value_19")]; + tensor var_393 = const()[name = tensor("op_393"), val = tensor(-2)]; + tensor var_394_interleave_0 = const()[name = tensor("op_394_interleave_0"), val = tensor(false)]; + tensor var_394 = concat(axis = var_393, interleave = var_394_interleave_0, values = key_states_21)[name = tensor("op_394")]; + tensor var_396 = const()[name = tensor("op_396"), val = tensor(-2)]; + tensor var_397_interleave_0 = const()[name = tensor("op_397_interleave_0"), val = tensor(false)]; + tensor var_397 = concat(axis = var_396, interleave = var_397_interleave_0, values = value_states_21)[name = tensor("op_397")]; + tensor var_415 = const()[name = tensor("op_415"), val = tensor(-2)]; + tensor key_23_interleave_0 = const()[name = tensor("key_23_interleave_0"), val = tensor(false)]; + tensor key_23 = concat(axis = var_415, interleave = key_23_interleave_0, values = key_states_23)[name = tensor("key_23")]; + tensor var_418 = const()[name = tensor("op_418"), val = tensor(-2)]; + tensor value_23_interleave_0 = const()[name = tensor("value_23_interleave_0"), val = tensor(false)]; + tensor value_23 = concat(axis = var_418, interleave = value_23_interleave_0, values = value_states_23)[name = tensor("value_23")]; + tensor var_437 = const()[name = tensor("op_437"), val = tensor(-2)]; + tensor var_438_interleave_0 = const()[name = tensor("op_438_interleave_0"), val = tensor(false)]; + tensor var_438 = concat(axis = var_437, interleave = var_438_interleave_0, values = key_states_25)[name = tensor("op_438")]; + tensor var_440 = const()[name = tensor("op_440"), val = tensor(-2)]; + tensor var_441_interleave_0 = const()[name = tensor("op_441_interleave_0"), val = tensor(false)]; + tensor var_441 = concat(axis = var_440, interleave = var_441_interleave_0, values = value_states_25)[name = tensor("op_441")]; + tensor var_459 = const()[name = tensor("op_459"), val = tensor(-2)]; + tensor key_27_interleave_0 = const()[name = tensor("key_27_interleave_0"), val = tensor(false)]; + tensor key_27 = concat(axis = var_459, interleave = key_27_interleave_0, values = key_states_27)[name = tensor("key_27")]; + tensor var_462 = const()[name = tensor("op_462"), val = tensor(-2)]; + tensor value_27_interleave_0 = const()[name = tensor("value_27_interleave_0"), val = tensor(false)]; + tensor value_27 = concat(axis = var_462, interleave = value_27_interleave_0, values = value_states_27)[name = tensor("value_27")]; + tensor var_481 = const()[name = tensor("op_481"), val = tensor(-2)]; + tensor var_482_interleave_0 = const()[name = tensor("op_482_interleave_0"), val = tensor(false)]; + tensor var_482 = concat(axis = var_481, interleave = var_482_interleave_0, values = key_states_29)[name = tensor("op_482")]; + tensor var_484 = const()[name = tensor("op_484"), val = tensor(-2)]; + tensor var_485_interleave_0 = const()[name = tensor("op_485_interleave_0"), val = tensor(false)]; + tensor var_485 = concat(axis = var_484, interleave = var_485_interleave_0, values = value_states_29)[name = tensor("op_485")]; + tensor var_503 = const()[name = tensor("op_503"), val = tensor(-2)]; + tensor key_31_interleave_0 = const()[name = tensor("key_31_interleave_0"), val = tensor(false)]; + tensor key_31 = concat(axis = var_503, interleave = key_31_interleave_0, values = key_states_31)[name = tensor("key_31")]; + tensor var_506 = const()[name = tensor("op_506"), val = tensor(-2)]; + tensor value_31_interleave_0 = const()[name = tensor("value_31_interleave_0"), val = tensor(false)]; + tensor value_31 = concat(axis = var_506, interleave = value_31_interleave_0, values = value_states_31)[name = tensor("value_31")]; + tensor var_525 = const()[name = tensor("op_525"), val = tensor(-2)]; + tensor var_526_interleave_0 = const()[name = tensor("op_526_interleave_0"), val = tensor(false)]; + tensor var_526 = concat(axis = var_525, interleave = var_526_interleave_0, values = key_states_33)[name = tensor("op_526")]; + tensor var_528 = const()[name = tensor("op_528"), val = tensor(-2)]; + tensor var_529_interleave_0 = const()[name = tensor("op_529_interleave_0"), val = tensor(false)]; + tensor var_529 = concat(axis = var_528, interleave = var_529_interleave_0, values = value_states_33)[name = tensor("op_529")]; + tensor var_547 = const()[name = tensor("op_547"), val = tensor(-2)]; + tensor key_35_interleave_0 = const()[name = tensor("key_35_interleave_0"), val = tensor(false)]; + tensor key_35 = concat(axis = var_547, interleave = key_35_interleave_0, values = key_states_35)[name = tensor("key_35")]; + tensor var_550 = const()[name = tensor("op_550"), val = tensor(-2)]; + tensor value_35_interleave_0 = const()[name = tensor("value_35_interleave_0"), val = tensor(false)]; + tensor value_35 = concat(axis = var_550, interleave = value_35_interleave_0, values = value_states_35)[name = tensor("value_35")]; + tensor var_569 = const()[name = tensor("op_569"), val = tensor(-2)]; + tensor var_570_interleave_0 = const()[name = tensor("op_570_interleave_0"), val = tensor(false)]; + tensor var_570 = concat(axis = var_569, interleave = var_570_interleave_0, values = key_states_37)[name = tensor("op_570")]; + tensor var_572 = const()[name = tensor("op_572"), val = tensor(-2)]; + tensor var_573_interleave_0 = const()[name = tensor("op_573_interleave_0"), val = tensor(false)]; + tensor var_573 = concat(axis = var_572, interleave = var_573_interleave_0, values = value_states_37)[name = tensor("op_573")]; + tensor var_591 = const()[name = tensor("op_591"), val = tensor(-2)]; + tensor key_39_interleave_0 = const()[name = tensor("key_39_interleave_0"), val = tensor(false)]; + tensor key_39 = concat(axis = var_591, interleave = key_39_interleave_0, values = key_states_39)[name = tensor("key_39")]; + tensor var_594 = const()[name = tensor("op_594"), val = tensor(-2)]; + tensor value_39_interleave_0 = const()[name = tensor("value_39_interleave_0"), val = tensor(false)]; + tensor value_39 = concat(axis = var_594, interleave = value_39_interleave_0, values = value_states_39)[name = tensor("value_39")]; + tensor var_613 = const()[name = tensor("op_613"), val = tensor(-2)]; + tensor var_614_interleave_0 = const()[name = tensor("op_614_interleave_0"), val = tensor(false)]; + tensor var_614 = concat(axis = var_613, interleave = var_614_interleave_0, values = key_states_41)[name = tensor("op_614")]; + tensor var_616 = const()[name = tensor("op_616"), val = tensor(-2)]; + tensor var_617_interleave_0 = const()[name = tensor("op_617_interleave_0"), val = tensor(false)]; + tensor var_617 = concat(axis = var_616, interleave = var_617_interleave_0, values = value_states_41)[name = tensor("op_617")]; + tensor var_635 = const()[name = tensor("op_635"), val = tensor(-2)]; + tensor key_43_interleave_0 = const()[name = tensor("key_43_interleave_0"), val = tensor(false)]; + tensor key_43 = concat(axis = var_635, interleave = key_43_interleave_0, values = key_states_43)[name = tensor("key_43")]; + tensor var_638 = const()[name = tensor("op_638"), val = tensor(-2)]; + tensor value_43_interleave_0 = const()[name = tensor("value_43_interleave_0"), val = tensor(false)]; + tensor value_43 = concat(axis = var_638, interleave = value_43_interleave_0, values = value_states_43)[name = tensor("value_43")]; + tensor var_657 = const()[name = tensor("op_657"), val = tensor(-2)]; + tensor var_658_interleave_0 = const()[name = tensor("op_658_interleave_0"), val = tensor(false)]; + tensor var_658 = concat(axis = var_657, interleave = var_658_interleave_0, values = key_states_45)[name = tensor("op_658")]; + tensor var_660 = const()[name = tensor("op_660"), val = tensor(-2)]; + tensor var_661_interleave_0 = const()[name = tensor("op_661_interleave_0"), val = tensor(false)]; + tensor var_661 = concat(axis = var_660, interleave = var_661_interleave_0, values = value_states_45)[name = tensor("op_661")]; + tensor var_679 = const()[name = tensor("op_679"), val = tensor(-2)]; + tensor key_interleave_0 = const()[name = tensor("key_interleave_0"), val = tensor(false)]; + tensor key = concat(axis = var_679, interleave = key_interleave_0, values = key_states_47)[name = tensor("key")]; + tensor var_682 = const()[name = tensor("op_682"), val = tensor(-2)]; + tensor value_interleave_0 = const()[name = tensor("value_interleave_0"), val = tensor(false)]; + tensor value = concat(axis = var_682, interleave = value_interleave_0, values = value_states_47)[name = tensor("value")]; + tensor var_685 = const()[name = tensor("op_685"), val = tensor(0x1.4f8b58p-17)]; + tensor var_687 = const()[name = tensor("op_687"), val = tensor(0x1p-3)]; + tensor var_689 = const()[name = tensor("op_689"), val = tensor(-2)]; + tensor var_698 = const()[name = tensor("op_698"), val = tensor(-0x1.fffffep+127)]; + tensor var_702 = const()[name = tensor("op_702"), val = tensor(0)]; + tensor var_705 = const()[name = tensor("op_705"), val = tensor(1)]; + tensor const_48 = const()[name = tensor("const_48"), val = tensor(1)]; + tensor var_737_axis_0 = const()[name = tensor("op_737_axis_0"), val = tensor(0)]; + tensor var_737_batch_dims_0 = const()[name = tensor("op_737_batch_dims_0"), val = tensor(0)]; + tensor var_737 = gather(axis = var_737_axis_0, batch_dims = var_737_batch_dims_0, indices = input_ids, x = decoder_embed_tokens_weight_palettized)[name = tensor("op_737")]; + tensor var_738 = const()[name = tensor("op_738"), val = tensor(0x1p+5)]; + tensor inputs_embeds = mul(x = var_737, y = var_738)[name = tensor("inputs_embeds")]; + tensor const_49 = const()[name = tensor("const_49"), val = tensor(1)]; + tensor seq_length = const()[name = tensor("seq_length"), val = tensor([1])]; + tensor var_743_shape = shape(x = var_174)[name = tensor("op_743_shape")]; + tensor gather_0_indices_0 = const()[name = tensor("gather_0_indices_0"), val = tensor(2)]; + tensor gather_0_axis_0 = const()[name = tensor("gather_0_axis_0"), val = tensor(0)]; + tensor gather_0_batch_dims_0 = const()[name = tensor("gather_0_batch_dims_0"), val = tensor(0)]; + tensor gather_0 = gather(axis = gather_0_axis_0, batch_dims = gather_0_batch_dims_0, indices = gather_0_indices_0, x = var_743_shape)[name = tensor("gather_0")]; + tensor var_745 = add(x = gather_0, y = seq_length)[name = tensor("op_745")]; + tensor var_746 = squeeze(x = var_745)[name = tensor("op_746")]; + tensor const_51 = const()[name = tensor("const_51"), val = tensor(1)]; + tensor cache_position = range_1d(end = var_746, start = gather_0, step = const_51)[name = tensor("cache_position")]; + tensor concat_0_axis_0 = const()[name = tensor("concat_0_axis_0"), val = tensor(0)]; + tensor concat_0_interleave_0 = const()[name = tensor("concat_0_interleave_0"), val = tensor(false)]; + tensor concat_0 = concat(axis = concat_0_axis_0, interleave = concat_0_interleave_0, values = (const_49, var_746))[name = tensor("concat_0")]; + tensor fill_0_value_0 = const()[name = tensor("fill_0_value_0"), val = tensor(0x1p+0)]; + tensor fill_0 = fill(shape = concat_0, value = fill_0_value_0)[name = tensor("fill_0")]; + tensor const_52 = const()[name = tensor("const_52"), val = tensor(1)]; + tensor var_753_shape = shape(x = fill_0)[name = tensor("op_753_shape")]; + tensor gather_1_indices_0 = const()[name = tensor("gather_1_indices_0"), val = tensor(1)]; + tensor gather_1_axis_0 = const()[name = tensor("gather_1_axis_0"), val = tensor(0)]; + tensor gather_1_batch_dims_0 = const()[name = tensor("gather_1_batch_dims_0"), val = tensor(0)]; + tensor gather_1 = gather(axis = gather_1_axis_0, batch_dims = gather_1_batch_dims_0, indices = gather_1_indices_0, x = var_753_shape)[name = tensor("gather_1")]; + tensor concat_1_axis_0 = const()[name = tensor("concat_1_axis_0"), val = tensor(0)]; + tensor concat_1_interleave_0 = const()[name = tensor("concat_1_interleave_0"), val = tensor(false)]; + tensor concat_1 = concat(axis = concat_1_axis_0, interleave = concat_1_interleave_0, values = (const_52, gather_1))[name = tensor("concat_1")]; + tensor causal_mask_1_value_0 = const()[name = tensor("causal_mask_1_value_0"), val = tensor(-0x1.fffffep+127)]; + tensor causal_mask_1 = fill(shape = concat_1, value = causal_mask_1_value_0)[name = tensor("causal_mask_1")]; + tensor const_54 = const()[name = tensor("const_54"), val = tensor(0)]; + tensor const_55 = const()[name = tensor("const_55"), val = tensor(1)]; + tensor var_757 = range_1d(end = gather_1, start = const_54, step = const_55)[name = tensor("op_757")]; + tensor var_758 = const()[name = tensor("op_758"), val = tensor([-1, 1])]; + tensor var_759 = reshape(shape = var_758, x = cache_position)[name = tensor("op_759")]; + tensor var_760 = greater(x = var_757, y = var_759)[name = tensor("op_760")]; + tensor var_760_promoted_dtype_0 = const()[name = tensor("op_760_promoted_dtype_0"), val = tensor("fp32")]; + tensor var_760_promoted = cast(dtype = var_760_promoted_dtype_0, x = var_760)[name = tensor("cast_3")]; + tensor causal_mask_3 = mul(x = causal_mask_1, y = var_760_promoted)[name = tensor("causal_mask_3")]; + tensor var_762_axes_0 = const()[name = tensor("op_762_axes_0"), val = tensor([0])]; + tensor var_762 = expand_dims(axes = var_762_axes_0, x = causal_mask_3)[name = tensor("op_762")]; + tensor var_763_axes_0 = const()[name = tensor("op_763_axes_0"), val = tensor([1])]; + tensor var_763 = expand_dims(axes = var_763_axes_0, x = var_762)[name = tensor("op_763")]; + tensor concat_2 = const()[name = tensor("concat_2"), val = tensor([1, 1, -1, -1])]; + tensor shape_0 = shape(x = var_763)[name = tensor("shape_0")]; + tensor equal_0 = const()[name = tensor("equal_0"), val = tensor([false, false, true, true])]; + tensor select_0 = select(a = shape_0, b = concat_2, cond = equal_0)[name = tensor("select_0")]; + tensor real_div_0 = real_div(x = select_0, y = shape_0)[name = tensor("real_div_0")]; + tensor causal_mask_5 = tile(reps = real_div_0, x = var_763)[name = tensor("causal_mask_5")]; + tensor concat_3_values0_0 = const()[name = tensor("concat_3_values0_0"), val = tensor(0)]; + tensor concat_3_values1_0 = const()[name = tensor("concat_3_values1_0"), val = tensor(0)]; + tensor concat_3_values2_0 = const()[name = tensor("concat_3_values2_0"), val = tensor(0)]; + tensor concat_3_axis_0 = const()[name = tensor("concat_3_axis_0"), val = tensor(0)]; + tensor concat_3_interleave_0 = const()[name = tensor("concat_3_interleave_0"), val = tensor(false)]; + tensor concat_3 = concat(axis = concat_3_axis_0, interleave = concat_3_interleave_0, values = (concat_3_values0_0, concat_3_values1_0, concat_3_values2_0, gather_1))[name = tensor("concat_3")]; + tensor var_773_begin_0 = const()[name = tensor("op_773_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_773_end_mask_0 = const()[name = tensor("op_773_end_mask_0"), val = tensor([true, true, true, false])]; + tensor var_773 = slice_by_index(begin = var_773_begin_0, end = concat_3, end_mask = var_773_end_mask_0, x = causal_mask_5)[name = tensor("op_773")]; + tensor var_775_axes_0 = const()[name = tensor("op_775_axes_0"), val = tensor([1])]; + tensor var_775 = expand_dims(axes = var_775_axes_0, x = fill_0)[name = tensor("op_775")]; + tensor var_776_axes_0 = const()[name = tensor("op_776_axes_0"), val = tensor([2])]; + tensor var_776 = expand_dims(axes = var_776_axes_0, x = var_775)[name = tensor("op_776")]; + tensor padding_mask_1 = add(x = var_773, y = var_776)[name = tensor("padding_mask_1")]; + tensor var_702_promoted = const()[name = tensor("op_702_promoted"), val = tensor(0x0p+0)]; + tensor padding_mask = equal(x = padding_mask_1, y = var_702_promoted)[name = tensor("padding_mask")]; + tensor var_785 = select(a = var_698, b = var_773, cond = padding_mask)[name = tensor("op_785")]; + tensor expand_dims_4_axes_0 = const()[name = tensor("expand_dims_4_axes_0"), val = tensor([0])]; + tensor expand_dims_4 = expand_dims(axes = expand_dims_4_axes_0, x = gather_1)[name = tensor("expand_dims_4")]; + tensor concat_6 = const()[name = tensor("concat_6"), val = tensor([0, 0, 0, 0])]; + tensor concat_7_values0_0 = const()[name = tensor("concat_7_values0_0"), val = tensor([0])]; + tensor concat_7_values1_0 = const()[name = tensor("concat_7_values1_0"), val = tensor([0])]; + tensor concat_7_values2_0 = const()[name = tensor("concat_7_values2_0"), val = tensor([0])]; + tensor concat_7_axis_0 = const()[name = tensor("concat_7_axis_0"), val = tensor(0)]; + tensor concat_7_interleave_0 = const()[name = tensor("concat_7_interleave_0"), val = tensor(false)]; + tensor concat_7 = concat(axis = concat_7_axis_0, interleave = concat_7_interleave_0, values = (concat_7_values0_0, concat_7_values1_0, concat_7_values2_0, expand_dims_4))[name = tensor("concat_7")]; + tensor causal_mask_internal_tensor_assign_1_stride_0 = const()[name = tensor("causal_mask_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor causal_mask_internal_tensor_assign_1_begin_mask_0 = const()[name = tensor("causal_mask_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor causal_mask_internal_tensor_assign_1_end_mask_0 = const()[name = tensor("causal_mask_internal_tensor_assign_1_end_mask_0"), val = tensor([true, true, true, false])]; + tensor causal_mask_internal_tensor_assign_1_squeeze_mask_0 = const()[name = tensor("causal_mask_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor shape_2 = shape(x = causal_mask_5)[name = tensor("shape_2")]; + tensor reduce_prod_0_keep_dims_0 = const()[name = tensor("reduce_prod_0_keep_dims_0"), val = tensor(false)]; + tensor reduce_prod_0 = reduce_prod(keep_dims = reduce_prod_0_keep_dims_0, x = shape_2)[name = tensor("reduce_prod_0")]; + tensor range_1d_0_start_0 = const()[name = tensor("range_1d_0_start_0"), val = tensor(0)]; + tensor range_1d_0_step_0 = const()[name = tensor("range_1d_0_step_0"), val = tensor(1)]; + tensor range_1d_0 = range_1d(end = reduce_prod_0, start = range_1d_0_start_0, step = range_1d_0_step_0)[name = tensor("range_1d_0")]; + tensor reshape_0 = reshape(shape = shape_2, x = range_1d_0)[name = tensor("reshape_0")]; + tensor slice_by_index_0 = slice_by_index(begin = concat_6, begin_mask = causal_mask_internal_tensor_assign_1_begin_mask_0, end = concat_7, end_mask = causal_mask_internal_tensor_assign_1_end_mask_0, squeeze_mask = causal_mask_internal_tensor_assign_1_squeeze_mask_0, stride = causal_mask_internal_tensor_assign_1_stride_0, x = reshape_0)[name = tensor("slice_by_index_0")]; + tensor reshape_1_shape_0 = const()[name = tensor("reshape_1_shape_0"), val = tensor([-1])]; + tensor reshape_1 = reshape(shape = reshape_1_shape_0, x = slice_by_index_0)[name = tensor("reshape_1")]; + tensor reshape_2_shape_0 = const()[name = tensor("reshape_2_shape_0"), val = tensor([-1])]; + tensor reshape_2 = reshape(shape = reshape_2_shape_0, x = var_785)[name = tensor("reshape_2")]; + tensor reshape_3_shape_0 = const()[name = tensor("reshape_3_shape_0"), val = tensor([-1])]; + tensor reshape_3 = reshape(shape = reshape_3_shape_0, x = causal_mask_5)[name = tensor("reshape_3")]; + tensor scatter_0_mode_0 = const()[name = tensor("scatter_0_mode_0"), val = tensor("update")]; + tensor scatter_0_axis_0 = const()[name = tensor("scatter_0_axis_0"), val = tensor(0)]; + tensor scatter_0 = scatter(axis = scatter_0_axis_0, data = reshape_3, indices = reshape_1, mode = scatter_0_mode_0, updates = reshape_2)[name = tensor("scatter_0")]; + tensor reshape_4 = reshape(shape = shape_2, x = scatter_0)[name = tensor("reshape_4")]; + tensor var_791_shape = shape(x = encoder_attention_mask)[name = tensor("op_791_shape")]; + tensor gather_3 = const()[name = tensor("gather_3"), val = tensor(1)]; + tensor gather_4_indices_0 = const()[name = tensor("gather_4_indices_0"), val = tensor(1)]; + tensor gather_4_axis_0 = const()[name = tensor("gather_4_axis_0"), val = tensor(0)]; + tensor gather_4_batch_dims_0 = const()[name = tensor("gather_4_batch_dims_0"), val = tensor(0)]; + tensor gather_4 = gather(axis = gather_4_axis_0, batch_dims = gather_4_batch_dims_0, indices = gather_4_indices_0, x = var_791_shape)[name = tensor("gather_4")]; + tensor var_794_axes_0 = const()[name = tensor("op_794_axes_0"), val = tensor([1])]; + tensor var_794 = expand_dims(axes = var_794_axes_0, x = encoder_attention_mask)[name = tensor("op_794")]; + tensor var_795_axes_0 = const()[name = tensor("op_795_axes_0"), val = tensor([2])]; + tensor var_795 = expand_dims(axes = var_795_axes_0, x = var_794)[name = tensor("op_795")]; + tensor concat_8_axis_0 = const()[name = tensor("concat_8_axis_0"), val = tensor(0)]; + tensor concat_8_interleave_0 = const()[name = tensor("concat_8_interleave_0"), val = tensor(false)]; + tensor concat_8 = concat(axis = concat_8_axis_0, interleave = concat_8_interleave_0, values = (gather_3, var_705, const_48, gather_4))[name = tensor("concat_8")]; + tensor shape_1 = shape(x = var_795)[name = tensor("shape_1")]; + tensor equal_1_y_0 = const()[name = tensor("equal_1_y_0"), val = tensor(-1)]; + tensor equal_1 = equal(x = concat_8, y = equal_1_y_0)[name = tensor("equal_1")]; + tensor select_1 = select(a = shape_1, b = concat_8, cond = equal_1)[name = tensor("select_1")]; + tensor real_div_1 = real_div(x = select_1, y = shape_1)[name = tensor("real_div_1")]; + tensor var_798 = tile(reps = real_div_1, x = var_795)[name = tensor("op_798")]; + tensor expanded_mask_dtype_0 = const()[name = tensor("expanded_mask_dtype_0"), val = tensor("fp32")]; + tensor const_56 = const()[name = tensor("const_56"), val = tensor(0x1p+0)]; + tensor expanded_mask = cast(dtype = expanded_mask_dtype_0, x = var_798)[name = tensor("cast_2")]; + tensor inverted_mask = sub(x = const_56, y = expanded_mask)[name = tensor("inverted_mask")]; + tensor var_803_dtype_0 = const()[name = tensor("op_803_dtype_0"), val = tensor("bool")]; + tensor var_803 = cast(dtype = var_803_dtype_0, x = inverted_mask)[name = tensor("cast_1")]; + tensor attention_mask_5 = select(a = var_698, b = inverted_mask, cond = var_803)[name = tensor("attention_mask_5")]; + tensor var_808 = not_equal(x = input_ids, y = var_705)[name = tensor("op_808")]; + tensor mask_dtype_0 = const()[name = tensor("mask_dtype_0"), val = tensor("int32")]; + tensor var_810_exclusive_0 = const()[name = tensor("op_810_exclusive_0"), val = tensor(false)]; + tensor var_810_reverse_0 = const()[name = tensor("op_810_reverse_0"), val = tensor(false)]; + tensor mask = cast(dtype = mask_dtype_0, x = var_808)[name = tensor("cast_0")]; + tensor var_810 = cumsum(axis = var_705, exclusive = var_810_exclusive_0, reverse = var_810_reverse_0, x = mask)[name = tensor("op_810")]; + tensor var_812 = add(x = var_810, y = gather_0)[name = tensor("op_812")]; + tensor incremental_indices = mul(x = var_812, y = mask)[name = tensor("incremental_indices")]; + tensor var_815 = const()[name = tensor("op_815"), val = tensor(1)]; + tensor var_816 = add(x = incremental_indices, y = var_815)[name = tensor("op_816")]; + tensor var_818 = const()[name = tensor("op_818"), val = tensor([-1])]; + tensor var_819 = reshape(shape = var_818, x = var_816)[name = tensor("op_819")]; + tensor var_820_batch_dims_0 = const()[name = tensor("op_820_batch_dims_0"), val = tensor(0)]; + tensor var_820 = gather(axis = var_702, batch_dims = var_820_batch_dims_0, indices = var_819, x = decoder_embed_positions_weights_palettized)[name = tensor("op_820")]; + tensor var_822 = const()[name = tensor("op_822"), val = tensor([1, 1, 1024])]; + tensor var_823 = reshape(shape = var_822, x = var_820)[name = tensor("op_823")]; + tensor input_3 = add(x = inputs_embeds, y = var_823)[name = tensor("input_3")]; + tensor hidden_states_1_axes_0 = const()[name = tensor("hidden_states_1_axes_0"), val = tensor([-1])]; + tensor hidden_states_1 = layer_norm(axes = hidden_states_1_axes_0, beta = decoder_layers_0_self_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_0_self_attn_layer_norm_weight, x = input_3)[name = tensor("hidden_states_1")]; + tensor var_847 = linear(bias = decoder_layers_0_self_attn_q_proj_bias, weight = decoder_layers_0_self_attn_q_proj_weight_palettized, x = hidden_states_1)[name = tensor("linear_0")]; + tensor var_848 = const()[name = tensor("op_848"), val = tensor([1, 1, -1, 64])]; + tensor var_849 = reshape(shape = var_848, x = var_847)[name = tensor("op_849")]; + tensor query_1_perm_0 = const()[name = tensor("query_1_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_49 = linear(bias = decoder_layers_0_self_attn_k_proj_bias, weight = decoder_layers_0_self_attn_k_proj_weight_palettized, x = hidden_states_1)[name = tensor("linear_1")]; + tensor value_states_49 = linear(bias = decoder_layers_0_self_attn_v_proj_bias, weight = decoder_layers_0_self_attn_v_proj_weight_palettized, x = hidden_states_1)[name = tensor("linear_2")]; + tensor var_857 = const()[name = tensor("op_857"), val = tensor([1, 1, -1, 64])]; + tensor var_858 = reshape(shape = var_857, x = key_states_49)[name = tensor("op_858")]; + tensor key_states_51_perm_0 = const()[name = tensor("key_states_51_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_860 = const()[name = tensor("op_860"), val = tensor([1, 1, -1, 64])]; + tensor var_861 = reshape(shape = var_860, x = value_states_49)[name = tensor("op_861")]; + tensor value_states_51_perm_0 = const()[name = tensor("value_states_51_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_1_interleave_0 = const()[name = tensor("key_1_interleave_0"), val = tensor(false)]; + tensor key_states_51 = transpose(perm = key_states_51_perm_0, x = var_858)[name = tensor("transpose_71")]; + tensor key_1 = concat(axis = var_689, interleave = key_1_interleave_0, values = (var_174, key_states_51))[name = tensor("key_1")]; + tensor value_1_interleave_0 = const()[name = tensor("value_1_interleave_0"), val = tensor(false)]; + tensor value_states_51 = transpose(perm = value_states_51_perm_0, x = var_861)[name = tensor("transpose_70")]; + tensor value_1 = concat(axis = var_689, interleave = value_1_interleave_0, values = (var_177, value_states_51))[name = tensor("value_1")]; + tensor var_867_shape = shape(x = key_1)[name = tensor("op_867_shape")]; + tensor gather_5_indices_0 = const()[name = tensor("gather_5_indices_0"), val = tensor(2)]; + tensor gather_5_axis_0 = const()[name = tensor("gather_5_axis_0"), val = tensor(0)]; + tensor gather_5_batch_dims_0 = const()[name = tensor("gather_5_batch_dims_0"), val = tensor(0)]; + tensor gather_5 = gather(axis = gather_5_axis_0, batch_dims = gather_5_batch_dims_0, indices = gather_5_indices_0, x = var_867_shape)[name = tensor("gather_5")]; + tensor concat_9_values0_0 = const()[name = tensor("concat_9_values0_0"), val = tensor(0)]; + tensor concat_9_values1_0 = const()[name = tensor("concat_9_values1_0"), val = tensor(0)]; + tensor concat_9_values2_0 = const()[name = tensor("concat_9_values2_0"), val = tensor(0)]; + tensor concat_9_axis_0 = const()[name = tensor("concat_9_axis_0"), val = tensor(0)]; + tensor concat_9_interleave_0 = const()[name = tensor("concat_9_interleave_0"), val = tensor(false)]; + tensor concat_9 = concat(axis = concat_9_axis_0, interleave = concat_9_interleave_0, values = (concat_9_values0_0, concat_9_values1_0, concat_9_values2_0, gather_5))[name = tensor("concat_9")]; + tensor attention_mask_3_begin_0 = const()[name = tensor("attention_mask_3_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_3_end_mask_0 = const()[name = tensor("attention_mask_3_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_3 = slice_by_index(begin = attention_mask_3_begin_0, end = concat_9, end_mask = attention_mask_3_end_mask_0, x = reshape_4)[name = tensor("attention_mask_3")]; + tensor query_1 = transpose(perm = query_1_perm_0, x = var_849)[name = tensor("transpose_69")]; + tensor mul_0 = mul(x = query_1, y = var_687)[name = tensor("mul_0")]; + tensor matmul_0_transpose_y_0 = const()[name = tensor("matmul_0_transpose_y_0"), val = tensor(true)]; + tensor matmul_0_transpose_x_0 = const()[name = tensor("matmul_0_transpose_x_0"), val = tensor(false)]; + tensor matmul_0 = matmul(transpose_x = matmul_0_transpose_x_0, transpose_y = matmul_0_transpose_y_0, x = mul_0, y = key_1)[name = tensor("matmul_0")]; + tensor add_0 = add(x = matmul_0, y = attention_mask_3)[name = tensor("add_0")]; + tensor softmax_0_axis_0 = const()[name = tensor("softmax_0_axis_0"), val = tensor(-1)]; + tensor softmax_0 = softmax(axis = softmax_0_axis_0, x = add_0)[name = tensor("softmax_0")]; + tensor attn_output_1_transpose_x_0 = const()[name = tensor("attn_output_1_transpose_x_0"), val = tensor(false)]; + tensor attn_output_1_transpose_y_0 = const()[name = tensor("attn_output_1_transpose_y_0"), val = tensor(false)]; + tensor attn_output_1 = matmul(transpose_x = attn_output_1_transpose_x_0, transpose_y = attn_output_1_transpose_y_0, x = softmax_0, y = value_1)[name = tensor("attn_output_1")]; + tensor var_873_perm_0 = const()[name = tensor("op_873_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_875 = const()[name = tensor("op_875"), val = tensor([1, 1, -1])]; + tensor var_873 = transpose(perm = var_873_perm_0, x = attn_output_1)[name = tensor("transpose_68")]; + tensor var_876 = reshape(shape = var_875, x = var_873)[name = tensor("op_876")]; + tensor input_9 = linear(bias = decoder_layers_0_self_attn_out_proj_bias, weight = decoder_layers_0_self_attn_out_proj_weight_palettized, x = var_876)[name = tensor("linear_3")]; + tensor input_11 = add(x = input_3, y = input_9)[name = tensor("input_11")]; + tensor hidden_states_5_axes_0 = const()[name = tensor("hidden_states_5_axes_0"), val = tensor([-1])]; + tensor hidden_states_5 = layer_norm(axes = hidden_states_5_axes_0, beta = decoder_layers_0_encoder_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_0_encoder_attn_layer_norm_weight, x = input_11)[name = tensor("hidden_states_5")]; + tensor var_897 = linear(bias = decoder_layers_0_encoder_attn_q_proj_bias, weight = decoder_layers_0_encoder_attn_q_proj_weight_palettized, x = hidden_states_5)[name = tensor("linear_4")]; + tensor var_898 = const()[name = tensor("op_898"), val = tensor([1, 1, -1, 64])]; + tensor var_899 = reshape(shape = var_898, x = var_897)[name = tensor("op_899")]; + tensor query_3_perm_0 = const()[name = tensor("query_3_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_901_shape = shape(x = key_3)[name = tensor("op_901_shape")]; + tensor gather_6_indices_0 = const()[name = tensor("gather_6_indices_0"), val = tensor(2)]; + tensor gather_6_axis_0 = const()[name = tensor("gather_6_axis_0"), val = tensor(0)]; + tensor gather_6_batch_dims_0 = const()[name = tensor("gather_6_batch_dims_0"), val = tensor(0)]; + tensor gather_6 = gather(axis = gather_6_axis_0, batch_dims = gather_6_batch_dims_0, indices = gather_6_indices_0, x = var_901_shape)[name = tensor("gather_6")]; + tensor concat_10_values0_0 = const()[name = tensor("concat_10_values0_0"), val = tensor(0)]; + tensor concat_10_values1_0 = const()[name = tensor("concat_10_values1_0"), val = tensor(0)]; + tensor concat_10_values2_0 = const()[name = tensor("concat_10_values2_0"), val = tensor(0)]; + tensor concat_10_axis_0 = const()[name = tensor("concat_10_axis_0"), val = tensor(0)]; + tensor concat_10_interleave_0 = const()[name = tensor("concat_10_interleave_0"), val = tensor(false)]; + tensor concat_10 = concat(axis = concat_10_axis_0, interleave = concat_10_interleave_0, values = (concat_10_values0_0, concat_10_values1_0, concat_10_values2_0, gather_6))[name = tensor("concat_10")]; + tensor attention_mask_7_begin_0 = const()[name = tensor("attention_mask_7_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_7_end_mask_0 = const()[name = tensor("attention_mask_7_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_7 = slice_by_index(begin = attention_mask_7_begin_0, end = concat_10, end_mask = attention_mask_7_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_7")]; + tensor query_3 = transpose(perm = query_3_perm_0, x = var_899)[name = tensor("transpose_67")]; + tensor mul_1 = mul(x = query_3, y = var_687)[name = tensor("mul_1")]; + tensor matmul_1_transpose_y_0 = const()[name = tensor("matmul_1_transpose_y_0"), val = tensor(true)]; + tensor matmul_1_transpose_x_0 = const()[name = tensor("matmul_1_transpose_x_0"), val = tensor(false)]; + tensor matmul_1 = matmul(transpose_x = matmul_1_transpose_x_0, transpose_y = matmul_1_transpose_y_0, x = mul_1, y = key_3)[name = tensor("matmul_1")]; + tensor add_1 = add(x = matmul_1, y = attention_mask_7)[name = tensor("add_1")]; + tensor softmax_1_axis_0 = const()[name = tensor("softmax_1_axis_0"), val = tensor(-1)]; + tensor softmax_1 = softmax(axis = softmax_1_axis_0, x = add_1)[name = tensor("softmax_1")]; + tensor attn_output_5_transpose_x_0 = const()[name = tensor("attn_output_5_transpose_x_0"), val = tensor(false)]; + tensor attn_output_5_transpose_y_0 = const()[name = tensor("attn_output_5_transpose_y_0"), val = tensor(false)]; + tensor attn_output_5 = matmul(transpose_x = attn_output_5_transpose_x_0, transpose_y = attn_output_5_transpose_y_0, x = softmax_1, y = value_3)[name = tensor("attn_output_5")]; + tensor var_907_perm_0 = const()[name = tensor("op_907_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_909 = const()[name = tensor("op_909"), val = tensor([1, 1, -1])]; + tensor var_907 = transpose(perm = var_907_perm_0, x = attn_output_5)[name = tensor("transpose_66")]; + tensor var_910 = reshape(shape = var_909, x = var_907)[name = tensor("op_910")]; + tensor input_15 = linear(bias = decoder_layers_0_encoder_attn_out_proj_bias, weight = decoder_layers_0_encoder_attn_out_proj_weight_palettized, x = var_910)[name = tensor("linear_5")]; + tensor input_17 = add(x = input_11, y = input_15)[name = tensor("input_17")]; + tensor input_19_axes_0 = const()[name = tensor("input_19_axes_0"), val = tensor([-1])]; + tensor input_19 = layer_norm(axes = input_19_axes_0, beta = decoder_layers_0_final_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_0_final_layer_norm_weight, x = input_17)[name = tensor("input_19")]; + tensor input_21 = linear(bias = decoder_layers_0_fc1_bias, weight = decoder_layers_0_fc1_weight_palettized, x = input_19)[name = tensor("linear_6")]; + tensor input_23 = relu(x = input_21)[name = tensor("input_23")]; + tensor input_27 = linear(bias = decoder_layers_0_fc2_bias, weight = decoder_layers_0_fc2_weight_palettized, x = input_23)[name = tensor("linear_7")]; + tensor input_29 = add(x = input_17, y = input_27)[name = tensor("input_29")]; + tensor hidden_states_11_axes_0 = const()[name = tensor("hidden_states_11_axes_0"), val = tensor([-1])]; + tensor hidden_states_11 = layer_norm(axes = hidden_states_11_axes_0, beta = decoder_layers_1_self_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_1_self_attn_layer_norm_weight, x = input_29)[name = tensor("hidden_states_11")]; + tensor var_954 = linear(bias = decoder_layers_1_self_attn_q_proj_bias, weight = decoder_layers_1_self_attn_q_proj_weight_palettized, x = hidden_states_11)[name = tensor("linear_8")]; + tensor var_955 = const()[name = tensor("op_955"), val = tensor([1, 1, -1, 64])]; + tensor var_956 = reshape(shape = var_955, x = var_954)[name = tensor("op_956")]; + tensor query_5_perm_0 = const()[name = tensor("query_5_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_53 = linear(bias = decoder_layers_1_self_attn_k_proj_bias, weight = decoder_layers_1_self_attn_k_proj_weight_palettized, x = hidden_states_11)[name = tensor("linear_9")]; + tensor value_states_53 = linear(bias = decoder_layers_1_self_attn_v_proj_bias, weight = decoder_layers_1_self_attn_v_proj_weight_palettized, x = hidden_states_11)[name = tensor("linear_10")]; + tensor var_964 = const()[name = tensor("op_964"), val = tensor([1, 1, -1, 64])]; + tensor var_965 = reshape(shape = var_964, x = key_states_53)[name = tensor("op_965")]; + tensor key_states_55_perm_0 = const()[name = tensor("key_states_55_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_967 = const()[name = tensor("op_967"), val = tensor([1, 1, -1, 64])]; + tensor var_968 = reshape(shape = var_967, x = value_states_53)[name = tensor("op_968")]; + tensor value_states_55_perm_0 = const()[name = tensor("value_states_55_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_5_interleave_0 = const()[name = tensor("key_5_interleave_0"), val = tensor(false)]; + tensor key_states_55 = transpose(perm = key_states_55_perm_0, x = var_965)[name = tensor("transpose_65")]; + tensor key_5 = concat(axis = var_689, interleave = key_5_interleave_0, values = (var_218, key_states_55))[name = tensor("key_5")]; + tensor value_5_interleave_0 = const()[name = tensor("value_5_interleave_0"), val = tensor(false)]; + tensor value_states_55 = transpose(perm = value_states_55_perm_0, x = var_968)[name = tensor("transpose_64")]; + tensor value_5 = concat(axis = var_689, interleave = value_5_interleave_0, values = (var_221, value_states_55))[name = tensor("value_5")]; + tensor var_974_shape = shape(x = key_5)[name = tensor("op_974_shape")]; + tensor gather_7_indices_0 = const()[name = tensor("gather_7_indices_0"), val = tensor(2)]; + tensor gather_7_axis_0 = const()[name = tensor("gather_7_axis_0"), val = tensor(0)]; + tensor gather_7_batch_dims_0 = const()[name = tensor("gather_7_batch_dims_0"), val = tensor(0)]; + tensor gather_7 = gather(axis = gather_7_axis_0, batch_dims = gather_7_batch_dims_0, indices = gather_7_indices_0, x = var_974_shape)[name = tensor("gather_7")]; + tensor concat_11_values0_0 = const()[name = tensor("concat_11_values0_0"), val = tensor(0)]; + tensor concat_11_values1_0 = const()[name = tensor("concat_11_values1_0"), val = tensor(0)]; + tensor concat_11_values2_0 = const()[name = tensor("concat_11_values2_0"), val = tensor(0)]; + tensor concat_11_axis_0 = const()[name = tensor("concat_11_axis_0"), val = tensor(0)]; + tensor concat_11_interleave_0 = const()[name = tensor("concat_11_interleave_0"), val = tensor(false)]; + tensor concat_11 = concat(axis = concat_11_axis_0, interleave = concat_11_interleave_0, values = (concat_11_values0_0, concat_11_values1_0, concat_11_values2_0, gather_7))[name = tensor("concat_11")]; + tensor attention_mask_9_begin_0 = const()[name = tensor("attention_mask_9_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_9_end_mask_0 = const()[name = tensor("attention_mask_9_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_9 = slice_by_index(begin = attention_mask_9_begin_0, end = concat_11, end_mask = attention_mask_9_end_mask_0, x = reshape_4)[name = tensor("attention_mask_9")]; + tensor query_5 = transpose(perm = query_5_perm_0, x = var_956)[name = tensor("transpose_63")]; + tensor mul_2 = mul(x = query_5, y = var_687)[name = tensor("mul_2")]; + tensor matmul_2_transpose_y_0 = const()[name = tensor("matmul_2_transpose_y_0"), val = tensor(true)]; + tensor matmul_2_transpose_x_0 = const()[name = tensor("matmul_2_transpose_x_0"), val = tensor(false)]; + tensor matmul_2 = matmul(transpose_x = matmul_2_transpose_x_0, transpose_y = matmul_2_transpose_y_0, x = mul_2, y = key_5)[name = tensor("matmul_2")]; + tensor add_2 = add(x = matmul_2, y = attention_mask_9)[name = tensor("add_2")]; + tensor softmax_2_axis_0 = const()[name = tensor("softmax_2_axis_0"), val = tensor(-1)]; + tensor softmax_2 = softmax(axis = softmax_2_axis_0, x = add_2)[name = tensor("softmax_2")]; + tensor attn_output_9_transpose_x_0 = const()[name = tensor("attn_output_9_transpose_x_0"), val = tensor(false)]; + tensor attn_output_9_transpose_y_0 = const()[name = tensor("attn_output_9_transpose_y_0"), val = tensor(false)]; + tensor attn_output_9 = matmul(transpose_x = attn_output_9_transpose_x_0, transpose_y = attn_output_9_transpose_y_0, x = softmax_2, y = value_5)[name = tensor("attn_output_9")]; + tensor var_980_perm_0 = const()[name = tensor("op_980_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_982 = const()[name = tensor("op_982"), val = tensor([1, 1, -1])]; + tensor var_980 = transpose(perm = var_980_perm_0, x = attn_output_9)[name = tensor("transpose_62")]; + tensor var_983 = reshape(shape = var_982, x = var_980)[name = tensor("op_983")]; + tensor input_33 = linear(bias = decoder_layers_1_self_attn_out_proj_bias, weight = decoder_layers_1_self_attn_out_proj_weight_palettized, x = var_983)[name = tensor("linear_11")]; + tensor input_35 = add(x = input_29, y = input_33)[name = tensor("input_35")]; + tensor hidden_states_15_axes_0 = const()[name = tensor("hidden_states_15_axes_0"), val = tensor([-1])]; + tensor hidden_states_15 = layer_norm(axes = hidden_states_15_axes_0, beta = decoder_layers_1_encoder_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_1_encoder_attn_layer_norm_weight, x = input_35)[name = tensor("hidden_states_15")]; + tensor var_1004 = linear(bias = decoder_layers_1_encoder_attn_q_proj_bias, weight = decoder_layers_1_encoder_attn_q_proj_weight_palettized, x = hidden_states_15)[name = tensor("linear_12")]; + tensor var_1005 = const()[name = tensor("op_1005"), val = tensor([1, 1, -1, 64])]; + tensor var_1006 = reshape(shape = var_1005, x = var_1004)[name = tensor("op_1006")]; + tensor query_7_perm_0 = const()[name = tensor("query_7_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1008_shape = shape(x = key_7)[name = tensor("op_1008_shape")]; + tensor gather_8_indices_0 = const()[name = tensor("gather_8_indices_0"), val = tensor(2)]; + tensor gather_8_axis_0 = const()[name = tensor("gather_8_axis_0"), val = tensor(0)]; + tensor gather_8_batch_dims_0 = const()[name = tensor("gather_8_batch_dims_0"), val = tensor(0)]; + tensor gather_8 = gather(axis = gather_8_axis_0, batch_dims = gather_8_batch_dims_0, indices = gather_8_indices_0, x = var_1008_shape)[name = tensor("gather_8")]; + tensor concat_12_values0_0 = const()[name = tensor("concat_12_values0_0"), val = tensor(0)]; + tensor concat_12_values1_0 = const()[name = tensor("concat_12_values1_0"), val = tensor(0)]; + tensor concat_12_values2_0 = const()[name = tensor("concat_12_values2_0"), val = tensor(0)]; + tensor concat_12_axis_0 = const()[name = tensor("concat_12_axis_0"), val = tensor(0)]; + tensor concat_12_interleave_0 = const()[name = tensor("concat_12_interleave_0"), val = tensor(false)]; + tensor concat_12 = concat(axis = concat_12_axis_0, interleave = concat_12_interleave_0, values = (concat_12_values0_0, concat_12_values1_0, concat_12_values2_0, gather_8))[name = tensor("concat_12")]; + tensor attention_mask_11_begin_0 = const()[name = tensor("attention_mask_11_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_11_end_mask_0 = const()[name = tensor("attention_mask_11_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_11 = slice_by_index(begin = attention_mask_11_begin_0, end = concat_12, end_mask = attention_mask_11_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_11")]; + tensor query_7 = transpose(perm = query_7_perm_0, x = var_1006)[name = tensor("transpose_61")]; + tensor mul_3 = mul(x = query_7, y = var_687)[name = tensor("mul_3")]; + tensor matmul_3_transpose_y_0 = const()[name = tensor("matmul_3_transpose_y_0"), val = tensor(true)]; + tensor matmul_3_transpose_x_0 = const()[name = tensor("matmul_3_transpose_x_0"), val = tensor(false)]; + tensor matmul_3 = matmul(transpose_x = matmul_3_transpose_x_0, transpose_y = matmul_3_transpose_y_0, x = mul_3, y = key_7)[name = tensor("matmul_3")]; + tensor add_3 = add(x = matmul_3, y = attention_mask_11)[name = tensor("add_3")]; + tensor softmax_3_axis_0 = const()[name = tensor("softmax_3_axis_0"), val = tensor(-1)]; + tensor softmax_3 = softmax(axis = softmax_3_axis_0, x = add_3)[name = tensor("softmax_3")]; + tensor attn_output_13_transpose_x_0 = const()[name = tensor("attn_output_13_transpose_x_0"), val = tensor(false)]; + tensor attn_output_13_transpose_y_0 = const()[name = tensor("attn_output_13_transpose_y_0"), val = tensor(false)]; + tensor attn_output_13 = matmul(transpose_x = attn_output_13_transpose_x_0, transpose_y = attn_output_13_transpose_y_0, x = softmax_3, y = value_7)[name = tensor("attn_output_13")]; + tensor var_1014_perm_0 = const()[name = tensor("op_1014_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1016 = const()[name = tensor("op_1016"), val = tensor([1, 1, -1])]; + tensor var_1014 = transpose(perm = var_1014_perm_0, x = attn_output_13)[name = tensor("transpose_60")]; + tensor var_1017 = reshape(shape = var_1016, x = var_1014)[name = tensor("op_1017")]; + tensor input_39 = linear(bias = decoder_layers_1_encoder_attn_out_proj_bias, weight = decoder_layers_1_encoder_attn_out_proj_weight_palettized, x = var_1017)[name = tensor("linear_13")]; + tensor input_41 = add(x = input_35, y = input_39)[name = tensor("input_41")]; + tensor input_43_axes_0 = const()[name = tensor("input_43_axes_0"), val = tensor([-1])]; + tensor input_43 = layer_norm(axes = input_43_axes_0, beta = decoder_layers_1_final_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_1_final_layer_norm_weight, x = input_41)[name = tensor("input_43")]; + tensor input_45 = linear(bias = decoder_layers_1_fc1_bias, weight = decoder_layers_1_fc1_weight_palettized, x = input_43)[name = tensor("linear_14")]; + tensor input_47 = relu(x = input_45)[name = tensor("input_47")]; + tensor input_51 = linear(bias = decoder_layers_1_fc2_bias, weight = decoder_layers_1_fc2_weight_palettized, x = input_47)[name = tensor("linear_15")]; + tensor input_53 = add(x = input_41, y = input_51)[name = tensor("input_53")]; + tensor hidden_states_21_axes_0 = const()[name = tensor("hidden_states_21_axes_0"), val = tensor([-1])]; + tensor hidden_states_21 = layer_norm(axes = hidden_states_21_axes_0, beta = decoder_layers_2_self_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_2_self_attn_layer_norm_weight, x = input_53)[name = tensor("hidden_states_21")]; + tensor var_1061 = linear(bias = decoder_layers_2_self_attn_q_proj_bias, weight = decoder_layers_2_self_attn_q_proj_weight_palettized, x = hidden_states_21)[name = tensor("linear_16")]; + tensor var_1062 = const()[name = tensor("op_1062"), val = tensor([1, 1, -1, 64])]; + tensor var_1063 = reshape(shape = var_1062, x = var_1061)[name = tensor("op_1063")]; + tensor query_9_perm_0 = const()[name = tensor("query_9_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_57 = linear(bias = decoder_layers_2_self_attn_k_proj_bias, weight = decoder_layers_2_self_attn_k_proj_weight_palettized, x = hidden_states_21)[name = tensor("linear_17")]; + tensor value_states_57 = linear(bias = decoder_layers_2_self_attn_v_proj_bias, weight = decoder_layers_2_self_attn_v_proj_weight_palettized, x = hidden_states_21)[name = tensor("linear_18")]; + tensor var_1071 = const()[name = tensor("op_1071"), val = tensor([1, 1, -1, 64])]; + tensor var_1072 = reshape(shape = var_1071, x = key_states_57)[name = tensor("op_1072")]; + tensor key_states_59_perm_0 = const()[name = tensor("key_states_59_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1074 = const()[name = tensor("op_1074"), val = tensor([1, 1, -1, 64])]; + tensor var_1075 = reshape(shape = var_1074, x = value_states_57)[name = tensor("op_1075")]; + tensor value_states_59_perm_0 = const()[name = tensor("value_states_59_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_9_interleave_0 = const()[name = tensor("key_9_interleave_0"), val = tensor(false)]; + tensor key_states_59 = transpose(perm = key_states_59_perm_0, x = var_1072)[name = tensor("transpose_59")]; + tensor key_9 = concat(axis = var_689, interleave = key_9_interleave_0, values = (var_262, key_states_59))[name = tensor("key_9")]; + tensor value_9_interleave_0 = const()[name = tensor("value_9_interleave_0"), val = tensor(false)]; + tensor value_states_59 = transpose(perm = value_states_59_perm_0, x = var_1075)[name = tensor("transpose_58")]; + tensor value_9 = concat(axis = var_689, interleave = value_9_interleave_0, values = (var_265, value_states_59))[name = tensor("value_9")]; + tensor var_1081_shape = shape(x = key_9)[name = tensor("op_1081_shape")]; + tensor gather_9_indices_0 = const()[name = tensor("gather_9_indices_0"), val = tensor(2)]; + tensor gather_9_axis_0 = const()[name = tensor("gather_9_axis_0"), val = tensor(0)]; + tensor gather_9_batch_dims_0 = const()[name = tensor("gather_9_batch_dims_0"), val = tensor(0)]; + tensor gather_9 = gather(axis = gather_9_axis_0, batch_dims = gather_9_batch_dims_0, indices = gather_9_indices_0, x = var_1081_shape)[name = tensor("gather_9")]; + tensor concat_13_values0_0 = const()[name = tensor("concat_13_values0_0"), val = tensor(0)]; + tensor concat_13_values1_0 = const()[name = tensor("concat_13_values1_0"), val = tensor(0)]; + tensor concat_13_values2_0 = const()[name = tensor("concat_13_values2_0"), val = tensor(0)]; + tensor concat_13_axis_0 = const()[name = tensor("concat_13_axis_0"), val = tensor(0)]; + tensor concat_13_interleave_0 = const()[name = tensor("concat_13_interleave_0"), val = tensor(false)]; + tensor concat_13 = concat(axis = concat_13_axis_0, interleave = concat_13_interleave_0, values = (concat_13_values0_0, concat_13_values1_0, concat_13_values2_0, gather_9))[name = tensor("concat_13")]; + tensor attention_mask_13_begin_0 = const()[name = tensor("attention_mask_13_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_13_end_mask_0 = const()[name = tensor("attention_mask_13_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_13 = slice_by_index(begin = attention_mask_13_begin_0, end = concat_13, end_mask = attention_mask_13_end_mask_0, x = reshape_4)[name = tensor("attention_mask_13")]; + tensor query_9 = transpose(perm = query_9_perm_0, x = var_1063)[name = tensor("transpose_57")]; + tensor mul_4 = mul(x = query_9, y = var_687)[name = tensor("mul_4")]; + tensor matmul_4_transpose_y_0 = const()[name = tensor("matmul_4_transpose_y_0"), val = tensor(true)]; + tensor matmul_4_transpose_x_0 = const()[name = tensor("matmul_4_transpose_x_0"), val = tensor(false)]; + tensor matmul_4 = matmul(transpose_x = matmul_4_transpose_x_0, transpose_y = matmul_4_transpose_y_0, x = mul_4, y = key_9)[name = tensor("matmul_4")]; + tensor add_4 = add(x = matmul_4, y = attention_mask_13)[name = tensor("add_4")]; + tensor softmax_4_axis_0 = const()[name = tensor("softmax_4_axis_0"), val = tensor(-1)]; + tensor softmax_4 = softmax(axis = softmax_4_axis_0, x = add_4)[name = tensor("softmax_4")]; + tensor attn_output_17_transpose_x_0 = const()[name = tensor("attn_output_17_transpose_x_0"), val = tensor(false)]; + tensor attn_output_17_transpose_y_0 = const()[name = tensor("attn_output_17_transpose_y_0"), val = tensor(false)]; + tensor attn_output_17 = matmul(transpose_x = attn_output_17_transpose_x_0, transpose_y = attn_output_17_transpose_y_0, x = softmax_4, y = value_9)[name = tensor("attn_output_17")]; + tensor var_1087_perm_0 = const()[name = tensor("op_1087_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1089 = const()[name = tensor("op_1089"), val = tensor([1, 1, -1])]; + tensor var_1087 = transpose(perm = var_1087_perm_0, x = attn_output_17)[name = tensor("transpose_56")]; + tensor var_1090 = reshape(shape = var_1089, x = var_1087)[name = tensor("op_1090")]; + tensor input_57 = linear(bias = decoder_layers_2_self_attn_out_proj_bias, weight = decoder_layers_2_self_attn_out_proj_weight_palettized, x = var_1090)[name = tensor("linear_19")]; + tensor input_59 = add(x = input_53, y = input_57)[name = tensor("input_59")]; + tensor hidden_states_25_axes_0 = const()[name = tensor("hidden_states_25_axes_0"), val = tensor([-1])]; + tensor hidden_states_25 = layer_norm(axes = hidden_states_25_axes_0, beta = decoder_layers_2_encoder_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_2_encoder_attn_layer_norm_weight, x = input_59)[name = tensor("hidden_states_25")]; + tensor var_1111 = linear(bias = decoder_layers_2_encoder_attn_q_proj_bias, weight = decoder_layers_2_encoder_attn_q_proj_weight_palettized, x = hidden_states_25)[name = tensor("linear_20")]; + tensor var_1112 = const()[name = tensor("op_1112"), val = tensor([1, 1, -1, 64])]; + tensor var_1113 = reshape(shape = var_1112, x = var_1111)[name = tensor("op_1113")]; + tensor query_11_perm_0 = const()[name = tensor("query_11_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1115_shape = shape(x = key_11)[name = tensor("op_1115_shape")]; + tensor gather_10_indices_0 = const()[name = tensor("gather_10_indices_0"), val = tensor(2)]; + tensor gather_10_axis_0 = const()[name = tensor("gather_10_axis_0"), val = tensor(0)]; + tensor gather_10_batch_dims_0 = const()[name = tensor("gather_10_batch_dims_0"), val = tensor(0)]; + tensor gather_10 = gather(axis = gather_10_axis_0, batch_dims = gather_10_batch_dims_0, indices = gather_10_indices_0, x = var_1115_shape)[name = tensor("gather_10")]; + tensor concat_14_values0_0 = const()[name = tensor("concat_14_values0_0"), val = tensor(0)]; + tensor concat_14_values1_0 = const()[name = tensor("concat_14_values1_0"), val = tensor(0)]; + tensor concat_14_values2_0 = const()[name = tensor("concat_14_values2_0"), val = tensor(0)]; + tensor concat_14_axis_0 = const()[name = tensor("concat_14_axis_0"), val = tensor(0)]; + tensor concat_14_interleave_0 = const()[name = tensor("concat_14_interleave_0"), val = tensor(false)]; + tensor concat_14 = concat(axis = concat_14_axis_0, interleave = concat_14_interleave_0, values = (concat_14_values0_0, concat_14_values1_0, concat_14_values2_0, gather_10))[name = tensor("concat_14")]; + tensor attention_mask_15_begin_0 = const()[name = tensor("attention_mask_15_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_15_end_mask_0 = const()[name = tensor("attention_mask_15_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_15 = slice_by_index(begin = attention_mask_15_begin_0, end = concat_14, end_mask = attention_mask_15_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_15")]; + tensor query_11 = transpose(perm = query_11_perm_0, x = var_1113)[name = tensor("transpose_55")]; + tensor mul_5 = mul(x = query_11, y = var_687)[name = tensor("mul_5")]; + tensor matmul_5_transpose_y_0 = const()[name = tensor("matmul_5_transpose_y_0"), val = tensor(true)]; + tensor matmul_5_transpose_x_0 = const()[name = tensor("matmul_5_transpose_x_0"), val = tensor(false)]; + tensor matmul_5 = matmul(transpose_x = matmul_5_transpose_x_0, transpose_y = matmul_5_transpose_y_0, x = mul_5, y = key_11)[name = tensor("matmul_5")]; + tensor add_5 = add(x = matmul_5, y = attention_mask_15)[name = tensor("add_5")]; + tensor softmax_5_axis_0 = const()[name = tensor("softmax_5_axis_0"), val = tensor(-1)]; + tensor softmax_5 = softmax(axis = softmax_5_axis_0, x = add_5)[name = tensor("softmax_5")]; + tensor attn_output_21_transpose_x_0 = const()[name = tensor("attn_output_21_transpose_x_0"), val = tensor(false)]; + tensor attn_output_21_transpose_y_0 = const()[name = tensor("attn_output_21_transpose_y_0"), val = tensor(false)]; + tensor attn_output_21 = matmul(transpose_x = attn_output_21_transpose_x_0, transpose_y = attn_output_21_transpose_y_0, x = softmax_5, y = value_11)[name = tensor("attn_output_21")]; + tensor var_1121_perm_0 = const()[name = tensor("op_1121_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1123 = const()[name = tensor("op_1123"), val = tensor([1, 1, -1])]; + tensor var_1121 = transpose(perm = var_1121_perm_0, x = attn_output_21)[name = tensor("transpose_54")]; + tensor var_1124 = reshape(shape = var_1123, x = var_1121)[name = tensor("op_1124")]; + tensor input_63 = linear(bias = decoder_layers_2_encoder_attn_out_proj_bias, weight = decoder_layers_2_encoder_attn_out_proj_weight_palettized, x = var_1124)[name = tensor("linear_21")]; + tensor input_65 = add(x = input_59, y = input_63)[name = tensor("input_65")]; + tensor input_67_axes_0 = const()[name = tensor("input_67_axes_0"), val = tensor([-1])]; + tensor input_67 = layer_norm(axes = input_67_axes_0, beta = decoder_layers_2_final_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_2_final_layer_norm_weight, x = input_65)[name = tensor("input_67")]; + tensor input_69 = linear(bias = decoder_layers_2_fc1_bias, weight = decoder_layers_2_fc1_weight_palettized, x = input_67)[name = tensor("linear_22")]; + tensor input_71 = relu(x = input_69)[name = tensor("input_71")]; + tensor input_75 = linear(bias = decoder_layers_2_fc2_bias, weight = decoder_layers_2_fc2_weight_palettized, x = input_71)[name = tensor("linear_23")]; + tensor input_77 = add(x = input_65, y = input_75)[name = tensor("input_77")]; + tensor hidden_states_31_axes_0 = const()[name = tensor("hidden_states_31_axes_0"), val = tensor([-1])]; + tensor hidden_states_31 = layer_norm(axes = hidden_states_31_axes_0, beta = decoder_layers_3_self_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_3_self_attn_layer_norm_weight, x = input_77)[name = tensor("hidden_states_31")]; + tensor var_1168 = linear(bias = decoder_layers_3_self_attn_q_proj_bias, weight = decoder_layers_3_self_attn_q_proj_weight_palettized, x = hidden_states_31)[name = tensor("linear_24")]; + tensor var_1169 = const()[name = tensor("op_1169"), val = tensor([1, 1, -1, 64])]; + tensor var_1170 = reshape(shape = var_1169, x = var_1168)[name = tensor("op_1170")]; + tensor query_13_perm_0 = const()[name = tensor("query_13_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_61 = linear(bias = decoder_layers_3_self_attn_k_proj_bias, weight = decoder_layers_3_self_attn_k_proj_weight_palettized, x = hidden_states_31)[name = tensor("linear_25")]; + tensor value_states_61 = linear(bias = decoder_layers_3_self_attn_v_proj_bias, weight = decoder_layers_3_self_attn_v_proj_weight_palettized, x = hidden_states_31)[name = tensor("linear_26")]; + tensor var_1178 = const()[name = tensor("op_1178"), val = tensor([1, 1, -1, 64])]; + tensor var_1179 = reshape(shape = var_1178, x = key_states_61)[name = tensor("op_1179")]; + tensor key_states_63_perm_0 = const()[name = tensor("key_states_63_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1181 = const()[name = tensor("op_1181"), val = tensor([1, 1, -1, 64])]; + tensor var_1182 = reshape(shape = var_1181, x = value_states_61)[name = tensor("op_1182")]; + tensor value_states_63_perm_0 = const()[name = tensor("value_states_63_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_13_interleave_0 = const()[name = tensor("key_13_interleave_0"), val = tensor(false)]; + tensor key_states_63 = transpose(perm = key_states_63_perm_0, x = var_1179)[name = tensor("transpose_53")]; + tensor key_13 = concat(axis = var_689, interleave = key_13_interleave_0, values = (var_306, key_states_63))[name = tensor("key_13")]; + tensor value_13_interleave_0 = const()[name = tensor("value_13_interleave_0"), val = tensor(false)]; + tensor value_states_63 = transpose(perm = value_states_63_perm_0, x = var_1182)[name = tensor("transpose_52")]; + tensor value_13 = concat(axis = var_689, interleave = value_13_interleave_0, values = (var_309, value_states_63))[name = tensor("value_13")]; + tensor var_1188_shape = shape(x = key_13)[name = tensor("op_1188_shape")]; + tensor gather_11_indices_0 = const()[name = tensor("gather_11_indices_0"), val = tensor(2)]; + tensor gather_11_axis_0 = const()[name = tensor("gather_11_axis_0"), val = tensor(0)]; + tensor gather_11_batch_dims_0 = const()[name = tensor("gather_11_batch_dims_0"), val = tensor(0)]; + tensor gather_11 = gather(axis = gather_11_axis_0, batch_dims = gather_11_batch_dims_0, indices = gather_11_indices_0, x = var_1188_shape)[name = tensor("gather_11")]; + tensor concat_15_values0_0 = const()[name = tensor("concat_15_values0_0"), val = tensor(0)]; + tensor concat_15_values1_0 = const()[name = tensor("concat_15_values1_0"), val = tensor(0)]; + tensor concat_15_values2_0 = const()[name = tensor("concat_15_values2_0"), val = tensor(0)]; + tensor concat_15_axis_0 = const()[name = tensor("concat_15_axis_0"), val = tensor(0)]; + tensor concat_15_interleave_0 = const()[name = tensor("concat_15_interleave_0"), val = tensor(false)]; + tensor concat_15 = concat(axis = concat_15_axis_0, interleave = concat_15_interleave_0, values = (concat_15_values0_0, concat_15_values1_0, concat_15_values2_0, gather_11))[name = tensor("concat_15")]; + tensor attention_mask_17_begin_0 = const()[name = tensor("attention_mask_17_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_17_end_mask_0 = const()[name = tensor("attention_mask_17_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_17 = slice_by_index(begin = attention_mask_17_begin_0, end = concat_15, end_mask = attention_mask_17_end_mask_0, x = reshape_4)[name = tensor("attention_mask_17")]; + tensor query_13 = transpose(perm = query_13_perm_0, x = var_1170)[name = tensor("transpose_51")]; + tensor mul_6 = mul(x = query_13, y = var_687)[name = tensor("mul_6")]; + tensor matmul_6_transpose_y_0 = const()[name = tensor("matmul_6_transpose_y_0"), val = tensor(true)]; + tensor matmul_6_transpose_x_0 = const()[name = tensor("matmul_6_transpose_x_0"), val = tensor(false)]; + tensor matmul_6 = matmul(transpose_x = matmul_6_transpose_x_0, transpose_y = matmul_6_transpose_y_0, x = mul_6, y = key_13)[name = tensor("matmul_6")]; + tensor add_6 = add(x = matmul_6, y = attention_mask_17)[name = tensor("add_6")]; + tensor softmax_6_axis_0 = const()[name = tensor("softmax_6_axis_0"), val = tensor(-1)]; + tensor softmax_6 = softmax(axis = softmax_6_axis_0, x = add_6)[name = tensor("softmax_6")]; + tensor attn_output_25_transpose_x_0 = const()[name = tensor("attn_output_25_transpose_x_0"), val = tensor(false)]; + tensor attn_output_25_transpose_y_0 = const()[name = tensor("attn_output_25_transpose_y_0"), val = tensor(false)]; + tensor attn_output_25 = matmul(transpose_x = attn_output_25_transpose_x_0, transpose_y = attn_output_25_transpose_y_0, x = softmax_6, y = value_13)[name = tensor("attn_output_25")]; + tensor var_1194_perm_0 = const()[name = tensor("op_1194_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1196 = const()[name = tensor("op_1196"), val = tensor([1, 1, -1])]; + tensor var_1194 = transpose(perm = var_1194_perm_0, x = attn_output_25)[name = tensor("transpose_50")]; + tensor var_1197 = reshape(shape = var_1196, x = var_1194)[name = tensor("op_1197")]; + tensor input_81 = linear(bias = decoder_layers_3_self_attn_out_proj_bias, weight = decoder_layers_3_self_attn_out_proj_weight_palettized, x = var_1197)[name = tensor("linear_27")]; + tensor input_83 = add(x = input_77, y = input_81)[name = tensor("input_83")]; + tensor hidden_states_35_axes_0 = const()[name = tensor("hidden_states_35_axes_0"), val = tensor([-1])]; + tensor hidden_states_35 = layer_norm(axes = hidden_states_35_axes_0, beta = decoder_layers_3_encoder_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_3_encoder_attn_layer_norm_weight, x = input_83)[name = tensor("hidden_states_35")]; + tensor var_1218 = linear(bias = decoder_layers_3_encoder_attn_q_proj_bias, weight = decoder_layers_3_encoder_attn_q_proj_weight_palettized, x = hidden_states_35)[name = tensor("linear_28")]; + tensor var_1219 = const()[name = tensor("op_1219"), val = tensor([1, 1, -1, 64])]; + tensor var_1220 = reshape(shape = var_1219, x = var_1218)[name = tensor("op_1220")]; + tensor query_15_perm_0 = const()[name = tensor("query_15_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1222_shape = shape(x = key_15)[name = tensor("op_1222_shape")]; + tensor gather_12_indices_0 = const()[name = tensor("gather_12_indices_0"), val = tensor(2)]; + tensor gather_12_axis_0 = const()[name = tensor("gather_12_axis_0"), val = tensor(0)]; + tensor gather_12_batch_dims_0 = const()[name = tensor("gather_12_batch_dims_0"), val = tensor(0)]; + tensor gather_12 = gather(axis = gather_12_axis_0, batch_dims = gather_12_batch_dims_0, indices = gather_12_indices_0, x = var_1222_shape)[name = tensor("gather_12")]; + tensor concat_16_values0_0 = const()[name = tensor("concat_16_values0_0"), val = tensor(0)]; + tensor concat_16_values1_0 = const()[name = tensor("concat_16_values1_0"), val = tensor(0)]; + tensor concat_16_values2_0 = const()[name = tensor("concat_16_values2_0"), val = tensor(0)]; + tensor concat_16_axis_0 = const()[name = tensor("concat_16_axis_0"), val = tensor(0)]; + tensor concat_16_interleave_0 = const()[name = tensor("concat_16_interleave_0"), val = tensor(false)]; + tensor concat_16 = concat(axis = concat_16_axis_0, interleave = concat_16_interleave_0, values = (concat_16_values0_0, concat_16_values1_0, concat_16_values2_0, gather_12))[name = tensor("concat_16")]; + tensor attention_mask_19_begin_0 = const()[name = tensor("attention_mask_19_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_19_end_mask_0 = const()[name = tensor("attention_mask_19_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_19 = slice_by_index(begin = attention_mask_19_begin_0, end = concat_16, end_mask = attention_mask_19_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_19")]; + tensor query_15 = transpose(perm = query_15_perm_0, x = var_1220)[name = tensor("transpose_49")]; + tensor mul_7 = mul(x = query_15, y = var_687)[name = tensor("mul_7")]; + tensor matmul_7_transpose_y_0 = const()[name = tensor("matmul_7_transpose_y_0"), val = tensor(true)]; + tensor matmul_7_transpose_x_0 = const()[name = tensor("matmul_7_transpose_x_0"), val = tensor(false)]; + tensor matmul_7 = matmul(transpose_x = matmul_7_transpose_x_0, transpose_y = matmul_7_transpose_y_0, x = mul_7, y = key_15)[name = tensor("matmul_7")]; + tensor add_7 = add(x = matmul_7, y = attention_mask_19)[name = tensor("add_7")]; + tensor softmax_7_axis_0 = const()[name = tensor("softmax_7_axis_0"), val = tensor(-1)]; + tensor softmax_7 = softmax(axis = softmax_7_axis_0, x = add_7)[name = tensor("softmax_7")]; + tensor attn_output_29_transpose_x_0 = const()[name = tensor("attn_output_29_transpose_x_0"), val = tensor(false)]; + tensor attn_output_29_transpose_y_0 = const()[name = tensor("attn_output_29_transpose_y_0"), val = tensor(false)]; + tensor attn_output_29 = matmul(transpose_x = attn_output_29_transpose_x_0, transpose_y = attn_output_29_transpose_y_0, x = softmax_7, y = value_15)[name = tensor("attn_output_29")]; + tensor var_1228_perm_0 = const()[name = tensor("op_1228_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1230 = const()[name = tensor("op_1230"), val = tensor([1, 1, -1])]; + tensor var_1228 = transpose(perm = var_1228_perm_0, x = attn_output_29)[name = tensor("transpose_48")]; + tensor var_1231 = reshape(shape = var_1230, x = var_1228)[name = tensor("op_1231")]; + tensor input_87 = linear(bias = decoder_layers_3_encoder_attn_out_proj_bias, weight = decoder_layers_3_encoder_attn_out_proj_weight_palettized, x = var_1231)[name = tensor("linear_29")]; + tensor input_89 = add(x = input_83, y = input_87)[name = tensor("input_89")]; + tensor input_91_axes_0 = const()[name = tensor("input_91_axes_0"), val = tensor([-1])]; + tensor input_91 = layer_norm(axes = input_91_axes_0, beta = decoder_layers_3_final_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_3_final_layer_norm_weight, x = input_89)[name = tensor("input_91")]; + tensor input_93 = linear(bias = decoder_layers_3_fc1_bias, weight = decoder_layers_3_fc1_weight_palettized, x = input_91)[name = tensor("linear_30")]; + tensor input_95 = relu(x = input_93)[name = tensor("input_95")]; + tensor input_99 = linear(bias = decoder_layers_3_fc2_bias, weight = decoder_layers_3_fc2_weight_palettized, x = input_95)[name = tensor("linear_31")]; + tensor input_101 = add(x = input_89, y = input_99)[name = tensor("input_101")]; + tensor hidden_states_41_axes_0 = const()[name = tensor("hidden_states_41_axes_0"), val = tensor([-1])]; + tensor hidden_states_41 = layer_norm(axes = hidden_states_41_axes_0, beta = decoder_layers_4_self_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_4_self_attn_layer_norm_weight, x = input_101)[name = tensor("hidden_states_41")]; + tensor var_1275 = linear(bias = decoder_layers_4_self_attn_q_proj_bias, weight = decoder_layers_4_self_attn_q_proj_weight_palettized, x = hidden_states_41)[name = tensor("linear_32")]; + tensor var_1276 = const()[name = tensor("op_1276"), val = tensor([1, 1, -1, 64])]; + tensor var_1277 = reshape(shape = var_1276, x = var_1275)[name = tensor("op_1277")]; + tensor query_17_perm_0 = const()[name = tensor("query_17_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_65 = linear(bias = decoder_layers_4_self_attn_k_proj_bias, weight = decoder_layers_4_self_attn_k_proj_weight_palettized, x = hidden_states_41)[name = tensor("linear_33")]; + tensor value_states_65 = linear(bias = decoder_layers_4_self_attn_v_proj_bias, weight = decoder_layers_4_self_attn_v_proj_weight_palettized, x = hidden_states_41)[name = tensor("linear_34")]; + tensor var_1285 = const()[name = tensor("op_1285"), val = tensor([1, 1, -1, 64])]; + tensor var_1286 = reshape(shape = var_1285, x = key_states_65)[name = tensor("op_1286")]; + tensor key_states_67_perm_0 = const()[name = tensor("key_states_67_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1288 = const()[name = tensor("op_1288"), val = tensor([1, 1, -1, 64])]; + tensor var_1289 = reshape(shape = var_1288, x = value_states_65)[name = tensor("op_1289")]; + tensor value_states_67_perm_0 = const()[name = tensor("value_states_67_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_17_interleave_0 = const()[name = tensor("key_17_interleave_0"), val = tensor(false)]; + tensor key_states_67 = transpose(perm = key_states_67_perm_0, x = var_1286)[name = tensor("transpose_47")]; + tensor key_17 = concat(axis = var_689, interleave = key_17_interleave_0, values = (var_350, key_states_67))[name = tensor("key_17")]; + tensor value_17_interleave_0 = const()[name = tensor("value_17_interleave_0"), val = tensor(false)]; + tensor value_states_67 = transpose(perm = value_states_67_perm_0, x = var_1289)[name = tensor("transpose_46")]; + tensor value_17 = concat(axis = var_689, interleave = value_17_interleave_0, values = (var_353, value_states_67))[name = tensor("value_17")]; + tensor var_1295_shape = shape(x = key_17)[name = tensor("op_1295_shape")]; + tensor gather_13_indices_0 = const()[name = tensor("gather_13_indices_0"), val = tensor(2)]; + tensor gather_13_axis_0 = const()[name = tensor("gather_13_axis_0"), val = tensor(0)]; + tensor gather_13_batch_dims_0 = const()[name = tensor("gather_13_batch_dims_0"), val = tensor(0)]; + tensor gather_13 = gather(axis = gather_13_axis_0, batch_dims = gather_13_batch_dims_0, indices = gather_13_indices_0, x = var_1295_shape)[name = tensor("gather_13")]; + tensor concat_17_values0_0 = const()[name = tensor("concat_17_values0_0"), val = tensor(0)]; + tensor concat_17_values1_0 = const()[name = tensor("concat_17_values1_0"), val = tensor(0)]; + tensor concat_17_values2_0 = const()[name = tensor("concat_17_values2_0"), val = tensor(0)]; + tensor concat_17_axis_0 = const()[name = tensor("concat_17_axis_0"), val = tensor(0)]; + tensor concat_17_interleave_0 = const()[name = tensor("concat_17_interleave_0"), val = tensor(false)]; + tensor concat_17 = concat(axis = concat_17_axis_0, interleave = concat_17_interleave_0, values = (concat_17_values0_0, concat_17_values1_0, concat_17_values2_0, gather_13))[name = tensor("concat_17")]; + tensor attention_mask_21_begin_0 = const()[name = tensor("attention_mask_21_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_21_end_mask_0 = const()[name = tensor("attention_mask_21_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_21 = slice_by_index(begin = attention_mask_21_begin_0, end = concat_17, end_mask = attention_mask_21_end_mask_0, x = reshape_4)[name = tensor("attention_mask_21")]; + tensor query_17 = transpose(perm = query_17_perm_0, x = var_1277)[name = tensor("transpose_45")]; + tensor mul_8 = mul(x = query_17, y = var_687)[name = tensor("mul_8")]; + tensor matmul_8_transpose_y_0 = const()[name = tensor("matmul_8_transpose_y_0"), val = tensor(true)]; + tensor matmul_8_transpose_x_0 = const()[name = tensor("matmul_8_transpose_x_0"), val = tensor(false)]; + tensor matmul_8 = matmul(transpose_x = matmul_8_transpose_x_0, transpose_y = matmul_8_transpose_y_0, x = mul_8, y = key_17)[name = tensor("matmul_8")]; + tensor add_8 = add(x = matmul_8, y = attention_mask_21)[name = tensor("add_8")]; + tensor softmax_8_axis_0 = const()[name = tensor("softmax_8_axis_0"), val = tensor(-1)]; + tensor softmax_8 = softmax(axis = softmax_8_axis_0, x = add_8)[name = tensor("softmax_8")]; + tensor attn_output_33_transpose_x_0 = const()[name = tensor("attn_output_33_transpose_x_0"), val = tensor(false)]; + tensor attn_output_33_transpose_y_0 = const()[name = tensor("attn_output_33_transpose_y_0"), val = tensor(false)]; + tensor attn_output_33 = matmul(transpose_x = attn_output_33_transpose_x_0, transpose_y = attn_output_33_transpose_y_0, x = softmax_8, y = value_17)[name = tensor("attn_output_33")]; + tensor var_1301_perm_0 = const()[name = tensor("op_1301_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1303 = const()[name = tensor("op_1303"), val = tensor([1, 1, -1])]; + tensor var_1301 = transpose(perm = var_1301_perm_0, x = attn_output_33)[name = tensor("transpose_44")]; + tensor var_1304 = reshape(shape = var_1303, x = var_1301)[name = tensor("op_1304")]; + tensor input_105 = linear(bias = decoder_layers_4_self_attn_out_proj_bias, weight = decoder_layers_4_self_attn_out_proj_weight_palettized, x = var_1304)[name = tensor("linear_35")]; + tensor input_107 = add(x = input_101, y = input_105)[name = tensor("input_107")]; + tensor hidden_states_45_axes_0 = const()[name = tensor("hidden_states_45_axes_0"), val = tensor([-1])]; + tensor hidden_states_45 = layer_norm(axes = hidden_states_45_axes_0, beta = decoder_layers_4_encoder_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_4_encoder_attn_layer_norm_weight, x = input_107)[name = tensor("hidden_states_45")]; + tensor var_1325 = linear(bias = decoder_layers_4_encoder_attn_q_proj_bias, weight = decoder_layers_4_encoder_attn_q_proj_weight_palettized, x = hidden_states_45)[name = tensor("linear_36")]; + tensor var_1326 = const()[name = tensor("op_1326"), val = tensor([1, 1, -1, 64])]; + tensor var_1327 = reshape(shape = var_1326, x = var_1325)[name = tensor("op_1327")]; + tensor query_19_perm_0 = const()[name = tensor("query_19_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1329_shape = shape(x = key_19)[name = tensor("op_1329_shape")]; + tensor gather_14_indices_0 = const()[name = tensor("gather_14_indices_0"), val = tensor(2)]; + tensor gather_14_axis_0 = const()[name = tensor("gather_14_axis_0"), val = tensor(0)]; + tensor gather_14_batch_dims_0 = const()[name = tensor("gather_14_batch_dims_0"), val = tensor(0)]; + tensor gather_14 = gather(axis = gather_14_axis_0, batch_dims = gather_14_batch_dims_0, indices = gather_14_indices_0, x = var_1329_shape)[name = tensor("gather_14")]; + tensor concat_18_values0_0 = const()[name = tensor("concat_18_values0_0"), val = tensor(0)]; + tensor concat_18_values1_0 = const()[name = tensor("concat_18_values1_0"), val = tensor(0)]; + tensor concat_18_values2_0 = const()[name = tensor("concat_18_values2_0"), val = tensor(0)]; + tensor concat_18_axis_0 = const()[name = tensor("concat_18_axis_0"), val = tensor(0)]; + tensor concat_18_interleave_0 = const()[name = tensor("concat_18_interleave_0"), val = tensor(false)]; + tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (concat_18_values0_0, concat_18_values1_0, concat_18_values2_0, gather_14))[name = tensor("concat_18")]; + tensor attention_mask_23_begin_0 = const()[name = tensor("attention_mask_23_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_23_end_mask_0 = const()[name = tensor("attention_mask_23_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_23 = slice_by_index(begin = attention_mask_23_begin_0, end = concat_18, end_mask = attention_mask_23_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_23")]; + tensor query_19 = transpose(perm = query_19_perm_0, x = var_1327)[name = tensor("transpose_43")]; + tensor mul_9 = mul(x = query_19, y = var_687)[name = tensor("mul_9")]; + tensor matmul_9_transpose_y_0 = const()[name = tensor("matmul_9_transpose_y_0"), val = tensor(true)]; + tensor matmul_9_transpose_x_0 = const()[name = tensor("matmul_9_transpose_x_0"), val = tensor(false)]; + tensor matmul_9 = matmul(transpose_x = matmul_9_transpose_x_0, transpose_y = matmul_9_transpose_y_0, x = mul_9, y = key_19)[name = tensor("matmul_9")]; + tensor add_9 = add(x = matmul_9, y = attention_mask_23)[name = tensor("add_9")]; + tensor softmax_9_axis_0 = const()[name = tensor("softmax_9_axis_0"), val = tensor(-1)]; + tensor softmax_9 = softmax(axis = softmax_9_axis_0, x = add_9)[name = tensor("softmax_9")]; + tensor attn_output_37_transpose_x_0 = const()[name = tensor("attn_output_37_transpose_x_0"), val = tensor(false)]; + tensor attn_output_37_transpose_y_0 = const()[name = tensor("attn_output_37_transpose_y_0"), val = tensor(false)]; + tensor attn_output_37 = matmul(transpose_x = attn_output_37_transpose_x_0, transpose_y = attn_output_37_transpose_y_0, x = softmax_9, y = value_19)[name = tensor("attn_output_37")]; + tensor var_1335_perm_0 = const()[name = tensor("op_1335_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1337 = const()[name = tensor("op_1337"), val = tensor([1, 1, -1])]; + tensor var_1335 = transpose(perm = var_1335_perm_0, x = attn_output_37)[name = tensor("transpose_42")]; + tensor var_1338 = reshape(shape = var_1337, x = var_1335)[name = tensor("op_1338")]; + tensor input_111 = linear(bias = decoder_layers_4_encoder_attn_out_proj_bias, weight = decoder_layers_4_encoder_attn_out_proj_weight_palettized, x = var_1338)[name = tensor("linear_37")]; + tensor input_113 = add(x = input_107, y = input_111)[name = tensor("input_113")]; + tensor input_115_axes_0 = const()[name = tensor("input_115_axes_0"), val = tensor([-1])]; + tensor input_115 = layer_norm(axes = input_115_axes_0, beta = decoder_layers_4_final_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_4_final_layer_norm_weight, x = input_113)[name = tensor("input_115")]; + tensor input_117 = linear(bias = decoder_layers_4_fc1_bias, weight = decoder_layers_4_fc1_weight_palettized, x = input_115)[name = tensor("linear_38")]; + tensor input_119 = relu(x = input_117)[name = tensor("input_119")]; + tensor input_123 = linear(bias = decoder_layers_4_fc2_bias, weight = decoder_layers_4_fc2_weight_palettized, x = input_119)[name = tensor("linear_39")]; + tensor input_125 = add(x = input_113, y = input_123)[name = tensor("input_125")]; + tensor hidden_states_51_axes_0 = const()[name = tensor("hidden_states_51_axes_0"), val = tensor([-1])]; + tensor hidden_states_51 = layer_norm(axes = hidden_states_51_axes_0, beta = decoder_layers_5_self_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_5_self_attn_layer_norm_weight, x = input_125)[name = tensor("hidden_states_51")]; + tensor var_1382 = linear(bias = decoder_layers_5_self_attn_q_proj_bias, weight = decoder_layers_5_self_attn_q_proj_weight_palettized, x = hidden_states_51)[name = tensor("linear_40")]; + tensor var_1383 = const()[name = tensor("op_1383"), val = tensor([1, 1, -1, 64])]; + tensor var_1384 = reshape(shape = var_1383, x = var_1382)[name = tensor("op_1384")]; + tensor query_21_perm_0 = const()[name = tensor("query_21_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_69 = linear(bias = decoder_layers_5_self_attn_k_proj_bias, weight = decoder_layers_5_self_attn_k_proj_weight_palettized, x = hidden_states_51)[name = tensor("linear_41")]; + tensor value_states_69 = linear(bias = decoder_layers_5_self_attn_v_proj_bias, weight = decoder_layers_5_self_attn_v_proj_weight_palettized, x = hidden_states_51)[name = tensor("linear_42")]; + tensor var_1392 = const()[name = tensor("op_1392"), val = tensor([1, 1, -1, 64])]; + tensor var_1393 = reshape(shape = var_1392, x = key_states_69)[name = tensor("op_1393")]; + tensor key_states_71_perm_0 = const()[name = tensor("key_states_71_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1395 = const()[name = tensor("op_1395"), val = tensor([1, 1, -1, 64])]; + tensor var_1396 = reshape(shape = var_1395, x = value_states_69)[name = tensor("op_1396")]; + tensor value_states_71_perm_0 = const()[name = tensor("value_states_71_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_21_interleave_0 = const()[name = tensor("key_21_interleave_0"), val = tensor(false)]; + tensor key_states_71 = transpose(perm = key_states_71_perm_0, x = var_1393)[name = tensor("transpose_41")]; + tensor key_21 = concat(axis = var_689, interleave = key_21_interleave_0, values = (var_394, key_states_71))[name = tensor("key_21")]; + tensor value_21_interleave_0 = const()[name = tensor("value_21_interleave_0"), val = tensor(false)]; + tensor value_states_71 = transpose(perm = value_states_71_perm_0, x = var_1396)[name = tensor("transpose_40")]; + tensor value_21 = concat(axis = var_689, interleave = value_21_interleave_0, values = (var_397, value_states_71))[name = tensor("value_21")]; + tensor var_1402_shape = shape(x = key_21)[name = tensor("op_1402_shape")]; + tensor gather_15_indices_0 = const()[name = tensor("gather_15_indices_0"), val = tensor(2)]; + tensor gather_15_axis_0 = const()[name = tensor("gather_15_axis_0"), val = tensor(0)]; + tensor gather_15_batch_dims_0 = const()[name = tensor("gather_15_batch_dims_0"), val = tensor(0)]; + tensor gather_15 = gather(axis = gather_15_axis_0, batch_dims = gather_15_batch_dims_0, indices = gather_15_indices_0, x = var_1402_shape)[name = tensor("gather_15")]; + tensor concat_19_values0_0 = const()[name = tensor("concat_19_values0_0"), val = tensor(0)]; + tensor concat_19_values1_0 = const()[name = tensor("concat_19_values1_0"), val = tensor(0)]; + tensor concat_19_values2_0 = const()[name = tensor("concat_19_values2_0"), val = tensor(0)]; + tensor concat_19_axis_0 = const()[name = tensor("concat_19_axis_0"), val = tensor(0)]; + tensor concat_19_interleave_0 = const()[name = tensor("concat_19_interleave_0"), val = tensor(false)]; + tensor concat_19 = concat(axis = concat_19_axis_0, interleave = concat_19_interleave_0, values = (concat_19_values0_0, concat_19_values1_0, concat_19_values2_0, gather_15))[name = tensor("concat_19")]; + tensor attention_mask_25_begin_0 = const()[name = tensor("attention_mask_25_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_25_end_mask_0 = const()[name = tensor("attention_mask_25_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_25 = slice_by_index(begin = attention_mask_25_begin_0, end = concat_19, end_mask = attention_mask_25_end_mask_0, x = reshape_4)[name = tensor("attention_mask_25")]; + tensor query_21 = transpose(perm = query_21_perm_0, x = var_1384)[name = tensor("transpose_39")]; + tensor mul_10 = mul(x = query_21, y = var_687)[name = tensor("mul_10")]; + tensor matmul_10_transpose_y_0 = const()[name = tensor("matmul_10_transpose_y_0"), val = tensor(true)]; + tensor matmul_10_transpose_x_0 = const()[name = tensor("matmul_10_transpose_x_0"), val = tensor(false)]; + tensor matmul_10 = matmul(transpose_x = matmul_10_transpose_x_0, transpose_y = matmul_10_transpose_y_0, x = mul_10, y = key_21)[name = tensor("matmul_10")]; + tensor add_10 = add(x = matmul_10, y = attention_mask_25)[name = tensor("add_10")]; + tensor softmax_10_axis_0 = const()[name = tensor("softmax_10_axis_0"), val = tensor(-1)]; + tensor softmax_10 = softmax(axis = softmax_10_axis_0, x = add_10)[name = tensor("softmax_10")]; + tensor attn_output_41_transpose_x_0 = const()[name = tensor("attn_output_41_transpose_x_0"), val = tensor(false)]; + tensor attn_output_41_transpose_y_0 = const()[name = tensor("attn_output_41_transpose_y_0"), val = tensor(false)]; + tensor attn_output_41 = matmul(transpose_x = attn_output_41_transpose_x_0, transpose_y = attn_output_41_transpose_y_0, x = softmax_10, y = value_21)[name = tensor("attn_output_41")]; + tensor var_1408_perm_0 = const()[name = tensor("op_1408_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1410 = const()[name = tensor("op_1410"), val = tensor([1, 1, -1])]; + tensor var_1408 = transpose(perm = var_1408_perm_0, x = attn_output_41)[name = tensor("transpose_38")]; + tensor var_1411 = reshape(shape = var_1410, x = var_1408)[name = tensor("op_1411")]; + tensor input_129 = linear(bias = decoder_layers_5_self_attn_out_proj_bias, weight = decoder_layers_5_self_attn_out_proj_weight_palettized, x = var_1411)[name = tensor("linear_43")]; + tensor input_131 = add(x = input_125, y = input_129)[name = tensor("input_131")]; + tensor hidden_states_55_axes_0 = const()[name = tensor("hidden_states_55_axes_0"), val = tensor([-1])]; + tensor hidden_states_55 = layer_norm(axes = hidden_states_55_axes_0, beta = decoder_layers_5_encoder_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_5_encoder_attn_layer_norm_weight, x = input_131)[name = tensor("hidden_states_55")]; + tensor var_1432 = linear(bias = decoder_layers_5_encoder_attn_q_proj_bias, weight = decoder_layers_5_encoder_attn_q_proj_weight_palettized, x = hidden_states_55)[name = tensor("linear_44")]; + tensor var_1433 = const()[name = tensor("op_1433"), val = tensor([1, 1, -1, 64])]; + tensor var_1434 = reshape(shape = var_1433, x = var_1432)[name = tensor("op_1434")]; + tensor query_23_perm_0 = const()[name = tensor("query_23_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1436_shape = shape(x = key_23)[name = tensor("op_1436_shape")]; + tensor gather_16_indices_0 = const()[name = tensor("gather_16_indices_0"), val = tensor(2)]; + tensor gather_16_axis_0 = const()[name = tensor("gather_16_axis_0"), val = tensor(0)]; + tensor gather_16_batch_dims_0 = const()[name = tensor("gather_16_batch_dims_0"), val = tensor(0)]; + tensor gather_16 = gather(axis = gather_16_axis_0, batch_dims = gather_16_batch_dims_0, indices = gather_16_indices_0, x = var_1436_shape)[name = tensor("gather_16")]; + tensor concat_20_values0_0 = const()[name = tensor("concat_20_values0_0"), val = tensor(0)]; + tensor concat_20_values1_0 = const()[name = tensor("concat_20_values1_0"), val = tensor(0)]; + tensor concat_20_values2_0 = const()[name = tensor("concat_20_values2_0"), val = tensor(0)]; + tensor concat_20_axis_0 = const()[name = tensor("concat_20_axis_0"), val = tensor(0)]; + tensor concat_20_interleave_0 = const()[name = tensor("concat_20_interleave_0"), val = tensor(false)]; + tensor concat_20 = concat(axis = concat_20_axis_0, interleave = concat_20_interleave_0, values = (concat_20_values0_0, concat_20_values1_0, concat_20_values2_0, gather_16))[name = tensor("concat_20")]; + tensor attention_mask_27_begin_0 = const()[name = tensor("attention_mask_27_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_27_end_mask_0 = const()[name = tensor("attention_mask_27_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_27 = slice_by_index(begin = attention_mask_27_begin_0, end = concat_20, end_mask = attention_mask_27_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_27")]; + tensor query_23 = transpose(perm = query_23_perm_0, x = var_1434)[name = tensor("transpose_37")]; + tensor mul_11 = mul(x = query_23, y = var_687)[name = tensor("mul_11")]; + tensor matmul_11_transpose_y_0 = const()[name = tensor("matmul_11_transpose_y_0"), val = tensor(true)]; + tensor matmul_11_transpose_x_0 = const()[name = tensor("matmul_11_transpose_x_0"), val = tensor(false)]; + tensor matmul_11 = matmul(transpose_x = matmul_11_transpose_x_0, transpose_y = matmul_11_transpose_y_0, x = mul_11, y = key_23)[name = tensor("matmul_11")]; + tensor add_11 = add(x = matmul_11, y = attention_mask_27)[name = tensor("add_11")]; + tensor softmax_11_axis_0 = const()[name = tensor("softmax_11_axis_0"), val = tensor(-1)]; + tensor softmax_11 = softmax(axis = softmax_11_axis_0, x = add_11)[name = tensor("softmax_11")]; + tensor attn_output_45_transpose_x_0 = const()[name = tensor("attn_output_45_transpose_x_0"), val = tensor(false)]; + tensor attn_output_45_transpose_y_0 = const()[name = tensor("attn_output_45_transpose_y_0"), val = tensor(false)]; + tensor attn_output_45 = matmul(transpose_x = attn_output_45_transpose_x_0, transpose_y = attn_output_45_transpose_y_0, x = softmax_11, y = value_23)[name = tensor("attn_output_45")]; + tensor var_1442_perm_0 = const()[name = tensor("op_1442_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1444 = const()[name = tensor("op_1444"), val = tensor([1, 1, -1])]; + tensor var_1442 = transpose(perm = var_1442_perm_0, x = attn_output_45)[name = tensor("transpose_36")]; + tensor var_1445 = reshape(shape = var_1444, x = var_1442)[name = tensor("op_1445")]; + tensor input_135 = linear(bias = decoder_layers_5_encoder_attn_out_proj_bias, weight = decoder_layers_5_encoder_attn_out_proj_weight_palettized, x = var_1445)[name = tensor("linear_45")]; + tensor input_137 = add(x = input_131, y = input_135)[name = tensor("input_137")]; + tensor input_139_axes_0 = const()[name = tensor("input_139_axes_0"), val = tensor([-1])]; + tensor input_139 = layer_norm(axes = input_139_axes_0, beta = decoder_layers_5_final_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_5_final_layer_norm_weight, x = input_137)[name = tensor("input_139")]; + tensor input_141 = linear(bias = decoder_layers_5_fc1_bias, weight = decoder_layers_5_fc1_weight_palettized, x = input_139)[name = tensor("linear_46")]; + tensor input_143 = relu(x = input_141)[name = tensor("input_143")]; + tensor input_147 = linear(bias = decoder_layers_5_fc2_bias, weight = decoder_layers_5_fc2_weight_palettized, x = input_143)[name = tensor("linear_47")]; + tensor input_149 = add(x = input_137, y = input_147)[name = tensor("input_149")]; + tensor hidden_states_61_axes_0 = const()[name = tensor("hidden_states_61_axes_0"), val = tensor([-1])]; + tensor hidden_states_61 = layer_norm(axes = hidden_states_61_axes_0, beta = decoder_layers_6_self_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_6_self_attn_layer_norm_weight, x = input_149)[name = tensor("hidden_states_61")]; + tensor var_1489 = linear(bias = decoder_layers_6_self_attn_q_proj_bias, weight = decoder_layers_6_self_attn_q_proj_weight_palettized, x = hidden_states_61)[name = tensor("linear_48")]; + tensor var_1490 = const()[name = tensor("op_1490"), val = tensor([1, 1, -1, 64])]; + tensor var_1491 = reshape(shape = var_1490, x = var_1489)[name = tensor("op_1491")]; + tensor query_25_perm_0 = const()[name = tensor("query_25_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_73 = linear(bias = decoder_layers_6_self_attn_k_proj_bias, weight = decoder_layers_6_self_attn_k_proj_weight_palettized, x = hidden_states_61)[name = tensor("linear_49")]; + tensor value_states_73 = linear(bias = decoder_layers_6_self_attn_v_proj_bias, weight = decoder_layers_6_self_attn_v_proj_weight_palettized, x = hidden_states_61)[name = tensor("linear_50")]; + tensor var_1499 = const()[name = tensor("op_1499"), val = tensor([1, 1, -1, 64])]; + tensor var_1500 = reshape(shape = var_1499, x = key_states_73)[name = tensor("op_1500")]; + tensor key_states_75_perm_0 = const()[name = tensor("key_states_75_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1502 = const()[name = tensor("op_1502"), val = tensor([1, 1, -1, 64])]; + tensor var_1503 = reshape(shape = var_1502, x = value_states_73)[name = tensor("op_1503")]; + tensor value_states_75_perm_0 = const()[name = tensor("value_states_75_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_25_interleave_0 = const()[name = tensor("key_25_interleave_0"), val = tensor(false)]; + tensor key_states_75 = transpose(perm = key_states_75_perm_0, x = var_1500)[name = tensor("transpose_35")]; + tensor key_25 = concat(axis = var_689, interleave = key_25_interleave_0, values = (var_438, key_states_75))[name = tensor("key_25")]; + tensor value_25_interleave_0 = const()[name = tensor("value_25_interleave_0"), val = tensor(false)]; + tensor value_states_75 = transpose(perm = value_states_75_perm_0, x = var_1503)[name = tensor("transpose_34")]; + tensor value_25 = concat(axis = var_689, interleave = value_25_interleave_0, values = (var_441, value_states_75))[name = tensor("value_25")]; + tensor var_1509_shape = shape(x = key_25)[name = tensor("op_1509_shape")]; + tensor gather_17_indices_0 = const()[name = tensor("gather_17_indices_0"), val = tensor(2)]; + tensor gather_17_axis_0 = const()[name = tensor("gather_17_axis_0"), val = tensor(0)]; + tensor gather_17_batch_dims_0 = const()[name = tensor("gather_17_batch_dims_0"), val = tensor(0)]; + tensor gather_17 = gather(axis = gather_17_axis_0, batch_dims = gather_17_batch_dims_0, indices = gather_17_indices_0, x = var_1509_shape)[name = tensor("gather_17")]; + tensor concat_21_values0_0 = const()[name = tensor("concat_21_values0_0"), val = tensor(0)]; + tensor concat_21_values1_0 = const()[name = tensor("concat_21_values1_0"), val = tensor(0)]; + tensor concat_21_values2_0 = const()[name = tensor("concat_21_values2_0"), val = tensor(0)]; + tensor concat_21_axis_0 = const()[name = tensor("concat_21_axis_0"), val = tensor(0)]; + tensor concat_21_interleave_0 = const()[name = tensor("concat_21_interleave_0"), val = tensor(false)]; + tensor concat_21 = concat(axis = concat_21_axis_0, interleave = concat_21_interleave_0, values = (concat_21_values0_0, concat_21_values1_0, concat_21_values2_0, gather_17))[name = tensor("concat_21")]; + tensor attention_mask_29_begin_0 = const()[name = tensor("attention_mask_29_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_29_end_mask_0 = const()[name = tensor("attention_mask_29_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_29 = slice_by_index(begin = attention_mask_29_begin_0, end = concat_21, end_mask = attention_mask_29_end_mask_0, x = reshape_4)[name = tensor("attention_mask_29")]; + tensor query_25 = transpose(perm = query_25_perm_0, x = var_1491)[name = tensor("transpose_33")]; + tensor mul_12 = mul(x = query_25, y = var_687)[name = tensor("mul_12")]; + tensor matmul_12_transpose_y_0 = const()[name = tensor("matmul_12_transpose_y_0"), val = tensor(true)]; + tensor matmul_12_transpose_x_0 = const()[name = tensor("matmul_12_transpose_x_0"), val = tensor(false)]; + tensor matmul_12 = matmul(transpose_x = matmul_12_transpose_x_0, transpose_y = matmul_12_transpose_y_0, x = mul_12, y = key_25)[name = tensor("matmul_12")]; + tensor add_12 = add(x = matmul_12, y = attention_mask_29)[name = tensor("add_12")]; + tensor softmax_12_axis_0 = const()[name = tensor("softmax_12_axis_0"), val = tensor(-1)]; + tensor softmax_12 = softmax(axis = softmax_12_axis_0, x = add_12)[name = tensor("softmax_12")]; + tensor attn_output_49_transpose_x_0 = const()[name = tensor("attn_output_49_transpose_x_0"), val = tensor(false)]; + tensor attn_output_49_transpose_y_0 = const()[name = tensor("attn_output_49_transpose_y_0"), val = tensor(false)]; + tensor attn_output_49 = matmul(transpose_x = attn_output_49_transpose_x_0, transpose_y = attn_output_49_transpose_y_0, x = softmax_12, y = value_25)[name = tensor("attn_output_49")]; + tensor var_1515_perm_0 = const()[name = tensor("op_1515_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1517 = const()[name = tensor("op_1517"), val = tensor([1, 1, -1])]; + tensor var_1515 = transpose(perm = var_1515_perm_0, x = attn_output_49)[name = tensor("transpose_32")]; + tensor var_1518 = reshape(shape = var_1517, x = var_1515)[name = tensor("op_1518")]; + tensor input_153 = linear(bias = decoder_layers_6_self_attn_out_proj_bias, weight = decoder_layers_6_self_attn_out_proj_weight_palettized, x = var_1518)[name = tensor("linear_51")]; + tensor input_155 = add(x = input_149, y = input_153)[name = tensor("input_155")]; + tensor hidden_states_65_axes_0 = const()[name = tensor("hidden_states_65_axes_0"), val = tensor([-1])]; + tensor hidden_states_65 = layer_norm(axes = hidden_states_65_axes_0, beta = decoder_layers_6_encoder_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_6_encoder_attn_layer_norm_weight, x = input_155)[name = tensor("hidden_states_65")]; + tensor var_1539 = linear(bias = decoder_layers_6_encoder_attn_q_proj_bias, weight = decoder_layers_6_encoder_attn_q_proj_weight_palettized, x = hidden_states_65)[name = tensor("linear_52")]; + tensor var_1540 = const()[name = tensor("op_1540"), val = tensor([1, 1, -1, 64])]; + tensor var_1541 = reshape(shape = var_1540, x = var_1539)[name = tensor("op_1541")]; + tensor query_27_perm_0 = const()[name = tensor("query_27_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1543_shape = shape(x = key_27)[name = tensor("op_1543_shape")]; + tensor gather_18_indices_0 = const()[name = tensor("gather_18_indices_0"), val = tensor(2)]; + tensor gather_18_axis_0 = const()[name = tensor("gather_18_axis_0"), val = tensor(0)]; + tensor gather_18_batch_dims_0 = const()[name = tensor("gather_18_batch_dims_0"), val = tensor(0)]; + tensor gather_18 = gather(axis = gather_18_axis_0, batch_dims = gather_18_batch_dims_0, indices = gather_18_indices_0, x = var_1543_shape)[name = tensor("gather_18")]; + tensor concat_22_values0_0 = const()[name = tensor("concat_22_values0_0"), val = tensor(0)]; + tensor concat_22_values1_0 = const()[name = tensor("concat_22_values1_0"), val = tensor(0)]; + tensor concat_22_values2_0 = const()[name = tensor("concat_22_values2_0"), val = tensor(0)]; + tensor concat_22_axis_0 = const()[name = tensor("concat_22_axis_0"), val = tensor(0)]; + tensor concat_22_interleave_0 = const()[name = tensor("concat_22_interleave_0"), val = tensor(false)]; + tensor concat_22 = concat(axis = concat_22_axis_0, interleave = concat_22_interleave_0, values = (concat_22_values0_0, concat_22_values1_0, concat_22_values2_0, gather_18))[name = tensor("concat_22")]; + tensor attention_mask_31_begin_0 = const()[name = tensor("attention_mask_31_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_31_end_mask_0 = const()[name = tensor("attention_mask_31_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_31 = slice_by_index(begin = attention_mask_31_begin_0, end = concat_22, end_mask = attention_mask_31_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_31")]; + tensor query_27 = transpose(perm = query_27_perm_0, x = var_1541)[name = tensor("transpose_31")]; + tensor mul_13 = mul(x = query_27, y = var_687)[name = tensor("mul_13")]; + tensor matmul_13_transpose_y_0 = const()[name = tensor("matmul_13_transpose_y_0"), val = tensor(true)]; + tensor matmul_13_transpose_x_0 = const()[name = tensor("matmul_13_transpose_x_0"), val = tensor(false)]; + tensor matmul_13 = matmul(transpose_x = matmul_13_transpose_x_0, transpose_y = matmul_13_transpose_y_0, x = mul_13, y = key_27)[name = tensor("matmul_13")]; + tensor add_13 = add(x = matmul_13, y = attention_mask_31)[name = tensor("add_13")]; + tensor softmax_13_axis_0 = const()[name = tensor("softmax_13_axis_0"), val = tensor(-1)]; + tensor softmax_13 = softmax(axis = softmax_13_axis_0, x = add_13)[name = tensor("softmax_13")]; + tensor attn_output_53_transpose_x_0 = const()[name = tensor("attn_output_53_transpose_x_0"), val = tensor(false)]; + tensor attn_output_53_transpose_y_0 = const()[name = tensor("attn_output_53_transpose_y_0"), val = tensor(false)]; + tensor attn_output_53 = matmul(transpose_x = attn_output_53_transpose_x_0, transpose_y = attn_output_53_transpose_y_0, x = softmax_13, y = value_27)[name = tensor("attn_output_53")]; + tensor var_1549_perm_0 = const()[name = tensor("op_1549_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1551 = const()[name = tensor("op_1551"), val = tensor([1, 1, -1])]; + tensor var_1549 = transpose(perm = var_1549_perm_0, x = attn_output_53)[name = tensor("transpose_30")]; + tensor var_1552 = reshape(shape = var_1551, x = var_1549)[name = tensor("op_1552")]; + tensor input_159 = linear(bias = decoder_layers_6_encoder_attn_out_proj_bias, weight = decoder_layers_6_encoder_attn_out_proj_weight_palettized, x = var_1552)[name = tensor("linear_53")]; + tensor input_161 = add(x = input_155, y = input_159)[name = tensor("input_161")]; + tensor input_163_axes_0 = const()[name = tensor("input_163_axes_0"), val = tensor([-1])]; + tensor input_163 = layer_norm(axes = input_163_axes_0, beta = decoder_layers_6_final_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_6_final_layer_norm_weight, x = input_161)[name = tensor("input_163")]; + tensor input_165 = linear(bias = decoder_layers_6_fc1_bias, weight = decoder_layers_6_fc1_weight_palettized, x = input_163)[name = tensor("linear_54")]; + tensor input_167 = relu(x = input_165)[name = tensor("input_167")]; + tensor input_171 = linear(bias = decoder_layers_6_fc2_bias, weight = decoder_layers_6_fc2_weight_palettized, x = input_167)[name = tensor("linear_55")]; + tensor input_173 = add(x = input_161, y = input_171)[name = tensor("input_173")]; + tensor hidden_states_71_axes_0 = const()[name = tensor("hidden_states_71_axes_0"), val = tensor([-1])]; + tensor hidden_states_71 = layer_norm(axes = hidden_states_71_axes_0, beta = decoder_layers_7_self_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_7_self_attn_layer_norm_weight, x = input_173)[name = tensor("hidden_states_71")]; + tensor var_1596 = linear(bias = decoder_layers_7_self_attn_q_proj_bias, weight = decoder_layers_7_self_attn_q_proj_weight_palettized, x = hidden_states_71)[name = tensor("linear_56")]; + tensor var_1597 = const()[name = tensor("op_1597"), val = tensor([1, 1, -1, 64])]; + tensor var_1598 = reshape(shape = var_1597, x = var_1596)[name = tensor("op_1598")]; + tensor query_29_perm_0 = const()[name = tensor("query_29_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_77 = linear(bias = decoder_layers_7_self_attn_k_proj_bias, weight = decoder_layers_7_self_attn_k_proj_weight_palettized, x = hidden_states_71)[name = tensor("linear_57")]; + tensor value_states_77 = linear(bias = decoder_layers_7_self_attn_v_proj_bias, weight = decoder_layers_7_self_attn_v_proj_weight_palettized, x = hidden_states_71)[name = tensor("linear_58")]; + tensor var_1606 = const()[name = tensor("op_1606"), val = tensor([1, 1, -1, 64])]; + tensor var_1607 = reshape(shape = var_1606, x = key_states_77)[name = tensor("op_1607")]; + tensor key_states_79_perm_0 = const()[name = tensor("key_states_79_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1609 = const()[name = tensor("op_1609"), val = tensor([1, 1, -1, 64])]; + tensor var_1610 = reshape(shape = var_1609, x = value_states_77)[name = tensor("op_1610")]; + tensor value_states_79_perm_0 = const()[name = tensor("value_states_79_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_29_interleave_0 = const()[name = tensor("key_29_interleave_0"), val = tensor(false)]; + tensor key_states_79 = transpose(perm = key_states_79_perm_0, x = var_1607)[name = tensor("transpose_29")]; + tensor key_29 = concat(axis = var_689, interleave = key_29_interleave_0, values = (var_482, key_states_79))[name = tensor("key_29")]; + tensor value_29_interleave_0 = const()[name = tensor("value_29_interleave_0"), val = tensor(false)]; + tensor value_states_79 = transpose(perm = value_states_79_perm_0, x = var_1610)[name = tensor("transpose_28")]; + tensor value_29 = concat(axis = var_689, interleave = value_29_interleave_0, values = (var_485, value_states_79))[name = tensor("value_29")]; + tensor var_1616_shape = shape(x = key_29)[name = tensor("op_1616_shape")]; + tensor gather_19_indices_0 = const()[name = tensor("gather_19_indices_0"), val = tensor(2)]; + tensor gather_19_axis_0 = const()[name = tensor("gather_19_axis_0"), val = tensor(0)]; + tensor gather_19_batch_dims_0 = const()[name = tensor("gather_19_batch_dims_0"), val = tensor(0)]; + tensor gather_19 = gather(axis = gather_19_axis_0, batch_dims = gather_19_batch_dims_0, indices = gather_19_indices_0, x = var_1616_shape)[name = tensor("gather_19")]; + tensor concat_23_values0_0 = const()[name = tensor("concat_23_values0_0"), val = tensor(0)]; + tensor concat_23_values1_0 = const()[name = tensor("concat_23_values1_0"), val = tensor(0)]; + tensor concat_23_values2_0 = const()[name = tensor("concat_23_values2_0"), val = tensor(0)]; + tensor concat_23_axis_0 = const()[name = tensor("concat_23_axis_0"), val = tensor(0)]; + tensor concat_23_interleave_0 = const()[name = tensor("concat_23_interleave_0"), val = tensor(false)]; + tensor concat_23 = concat(axis = concat_23_axis_0, interleave = concat_23_interleave_0, values = (concat_23_values0_0, concat_23_values1_0, concat_23_values2_0, gather_19))[name = tensor("concat_23")]; + tensor attention_mask_33_begin_0 = const()[name = tensor("attention_mask_33_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_33_end_mask_0 = const()[name = tensor("attention_mask_33_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_33 = slice_by_index(begin = attention_mask_33_begin_0, end = concat_23, end_mask = attention_mask_33_end_mask_0, x = reshape_4)[name = tensor("attention_mask_33")]; + tensor query_29 = transpose(perm = query_29_perm_0, x = var_1598)[name = tensor("transpose_27")]; + tensor mul_14 = mul(x = query_29, y = var_687)[name = tensor("mul_14")]; + tensor matmul_14_transpose_y_0 = const()[name = tensor("matmul_14_transpose_y_0"), val = tensor(true)]; + tensor matmul_14_transpose_x_0 = const()[name = tensor("matmul_14_transpose_x_0"), val = tensor(false)]; + tensor matmul_14 = matmul(transpose_x = matmul_14_transpose_x_0, transpose_y = matmul_14_transpose_y_0, x = mul_14, y = key_29)[name = tensor("matmul_14")]; + tensor add_14 = add(x = matmul_14, y = attention_mask_33)[name = tensor("add_14")]; + tensor softmax_14_axis_0 = const()[name = tensor("softmax_14_axis_0"), val = tensor(-1)]; + tensor softmax_14 = softmax(axis = softmax_14_axis_0, x = add_14)[name = tensor("softmax_14")]; + tensor attn_output_57_transpose_x_0 = const()[name = tensor("attn_output_57_transpose_x_0"), val = tensor(false)]; + tensor attn_output_57_transpose_y_0 = const()[name = tensor("attn_output_57_transpose_y_0"), val = tensor(false)]; + tensor attn_output_57 = matmul(transpose_x = attn_output_57_transpose_x_0, transpose_y = attn_output_57_transpose_y_0, x = softmax_14, y = value_29)[name = tensor("attn_output_57")]; + tensor var_1622_perm_0 = const()[name = tensor("op_1622_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1624 = const()[name = tensor("op_1624"), val = tensor([1, 1, -1])]; + tensor var_1622 = transpose(perm = var_1622_perm_0, x = attn_output_57)[name = tensor("transpose_26")]; + tensor var_1625 = reshape(shape = var_1624, x = var_1622)[name = tensor("op_1625")]; + tensor input_177 = linear(bias = decoder_layers_7_self_attn_out_proj_bias, weight = decoder_layers_7_self_attn_out_proj_weight_palettized, x = var_1625)[name = tensor("linear_59")]; + tensor input_179 = add(x = input_173, y = input_177)[name = tensor("input_179")]; + tensor hidden_states_75_axes_0 = const()[name = tensor("hidden_states_75_axes_0"), val = tensor([-1])]; + tensor hidden_states_75 = layer_norm(axes = hidden_states_75_axes_0, beta = decoder_layers_7_encoder_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_7_encoder_attn_layer_norm_weight, x = input_179)[name = tensor("hidden_states_75")]; + tensor var_1646 = linear(bias = decoder_layers_7_encoder_attn_q_proj_bias, weight = decoder_layers_7_encoder_attn_q_proj_weight_palettized, x = hidden_states_75)[name = tensor("linear_60")]; + tensor var_1647 = const()[name = tensor("op_1647"), val = tensor([1, 1, -1, 64])]; + tensor var_1648 = reshape(shape = var_1647, x = var_1646)[name = tensor("op_1648")]; + tensor query_31_perm_0 = const()[name = tensor("query_31_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1650_shape = shape(x = key_31)[name = tensor("op_1650_shape")]; + tensor gather_20_indices_0 = const()[name = tensor("gather_20_indices_0"), val = tensor(2)]; + tensor gather_20_axis_0 = const()[name = tensor("gather_20_axis_0"), val = tensor(0)]; + tensor gather_20_batch_dims_0 = const()[name = tensor("gather_20_batch_dims_0"), val = tensor(0)]; + tensor gather_20 = gather(axis = gather_20_axis_0, batch_dims = gather_20_batch_dims_0, indices = gather_20_indices_0, x = var_1650_shape)[name = tensor("gather_20")]; + tensor concat_24_values0_0 = const()[name = tensor("concat_24_values0_0"), val = tensor(0)]; + tensor concat_24_values1_0 = const()[name = tensor("concat_24_values1_0"), val = tensor(0)]; + tensor concat_24_values2_0 = const()[name = tensor("concat_24_values2_0"), val = tensor(0)]; + tensor concat_24_axis_0 = const()[name = tensor("concat_24_axis_0"), val = tensor(0)]; + tensor concat_24_interleave_0 = const()[name = tensor("concat_24_interleave_0"), val = tensor(false)]; + tensor concat_24 = concat(axis = concat_24_axis_0, interleave = concat_24_interleave_0, values = (concat_24_values0_0, concat_24_values1_0, concat_24_values2_0, gather_20))[name = tensor("concat_24")]; + tensor attention_mask_35_begin_0 = const()[name = tensor("attention_mask_35_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_35_end_mask_0 = const()[name = tensor("attention_mask_35_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_35 = slice_by_index(begin = attention_mask_35_begin_0, end = concat_24, end_mask = attention_mask_35_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_35")]; + tensor query_31 = transpose(perm = query_31_perm_0, x = var_1648)[name = tensor("transpose_25")]; + tensor mul_15 = mul(x = query_31, y = var_687)[name = tensor("mul_15")]; + tensor matmul_15_transpose_y_0 = const()[name = tensor("matmul_15_transpose_y_0"), val = tensor(true)]; + tensor matmul_15_transpose_x_0 = const()[name = tensor("matmul_15_transpose_x_0"), val = tensor(false)]; + tensor matmul_15 = matmul(transpose_x = matmul_15_transpose_x_0, transpose_y = matmul_15_transpose_y_0, x = mul_15, y = key_31)[name = tensor("matmul_15")]; + tensor add_15 = add(x = matmul_15, y = attention_mask_35)[name = tensor("add_15")]; + tensor softmax_15_axis_0 = const()[name = tensor("softmax_15_axis_0"), val = tensor(-1)]; + tensor softmax_15 = softmax(axis = softmax_15_axis_0, x = add_15)[name = tensor("softmax_15")]; + tensor attn_output_61_transpose_x_0 = const()[name = tensor("attn_output_61_transpose_x_0"), val = tensor(false)]; + tensor attn_output_61_transpose_y_0 = const()[name = tensor("attn_output_61_transpose_y_0"), val = tensor(false)]; + tensor attn_output_61 = matmul(transpose_x = attn_output_61_transpose_x_0, transpose_y = attn_output_61_transpose_y_0, x = softmax_15, y = value_31)[name = tensor("attn_output_61")]; + tensor var_1656_perm_0 = const()[name = tensor("op_1656_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1658 = const()[name = tensor("op_1658"), val = tensor([1, 1, -1])]; + tensor var_1656 = transpose(perm = var_1656_perm_0, x = attn_output_61)[name = tensor("transpose_24")]; + tensor var_1659 = reshape(shape = var_1658, x = var_1656)[name = tensor("op_1659")]; + tensor input_183 = linear(bias = decoder_layers_7_encoder_attn_out_proj_bias, weight = decoder_layers_7_encoder_attn_out_proj_weight_palettized, x = var_1659)[name = tensor("linear_61")]; + tensor input_185 = add(x = input_179, y = input_183)[name = tensor("input_185")]; + tensor input_187_axes_0 = const()[name = tensor("input_187_axes_0"), val = tensor([-1])]; + tensor input_187 = layer_norm(axes = input_187_axes_0, beta = decoder_layers_7_final_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_7_final_layer_norm_weight, x = input_185)[name = tensor("input_187")]; + tensor input_189 = linear(bias = decoder_layers_7_fc1_bias, weight = decoder_layers_7_fc1_weight_palettized, x = input_187)[name = tensor("linear_62")]; + tensor input_191 = relu(x = input_189)[name = tensor("input_191")]; + tensor input_195 = linear(bias = decoder_layers_7_fc2_bias, weight = decoder_layers_7_fc2_weight_palettized, x = input_191)[name = tensor("linear_63")]; + tensor input_197 = add(x = input_185, y = input_195)[name = tensor("input_197")]; + tensor hidden_states_81_axes_0 = const()[name = tensor("hidden_states_81_axes_0"), val = tensor([-1])]; + tensor hidden_states_81 = layer_norm(axes = hidden_states_81_axes_0, beta = decoder_layers_8_self_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_8_self_attn_layer_norm_weight, x = input_197)[name = tensor("hidden_states_81")]; + tensor var_1703 = linear(bias = decoder_layers_8_self_attn_q_proj_bias, weight = decoder_layers_8_self_attn_q_proj_weight_palettized, x = hidden_states_81)[name = tensor("linear_64")]; + tensor var_1704 = const()[name = tensor("op_1704"), val = tensor([1, 1, -1, 64])]; + tensor var_1705 = reshape(shape = var_1704, x = var_1703)[name = tensor("op_1705")]; + tensor query_33_perm_0 = const()[name = tensor("query_33_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_81 = linear(bias = decoder_layers_8_self_attn_k_proj_bias, weight = decoder_layers_8_self_attn_k_proj_weight_palettized, x = hidden_states_81)[name = tensor("linear_65")]; + tensor value_states_81 = linear(bias = decoder_layers_8_self_attn_v_proj_bias, weight = decoder_layers_8_self_attn_v_proj_weight_palettized, x = hidden_states_81)[name = tensor("linear_66")]; + tensor var_1713 = const()[name = tensor("op_1713"), val = tensor([1, 1, -1, 64])]; + tensor var_1714 = reshape(shape = var_1713, x = key_states_81)[name = tensor("op_1714")]; + tensor key_states_83_perm_0 = const()[name = tensor("key_states_83_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1716 = const()[name = tensor("op_1716"), val = tensor([1, 1, -1, 64])]; + tensor var_1717 = reshape(shape = var_1716, x = value_states_81)[name = tensor("op_1717")]; + tensor value_states_83_perm_0 = const()[name = tensor("value_states_83_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_33_interleave_0 = const()[name = tensor("key_33_interleave_0"), val = tensor(false)]; + tensor key_states_83 = transpose(perm = key_states_83_perm_0, x = var_1714)[name = tensor("transpose_23")]; + tensor key_33 = concat(axis = var_689, interleave = key_33_interleave_0, values = (var_526, key_states_83))[name = tensor("key_33")]; + tensor value_33_interleave_0 = const()[name = tensor("value_33_interleave_0"), val = tensor(false)]; + tensor value_states_83 = transpose(perm = value_states_83_perm_0, x = var_1717)[name = tensor("transpose_22")]; + tensor value_33 = concat(axis = var_689, interleave = value_33_interleave_0, values = (var_529, value_states_83))[name = tensor("value_33")]; + tensor var_1723_shape = shape(x = key_33)[name = tensor("op_1723_shape")]; + tensor gather_21_indices_0 = const()[name = tensor("gather_21_indices_0"), val = tensor(2)]; + tensor gather_21_axis_0 = const()[name = tensor("gather_21_axis_0"), val = tensor(0)]; + tensor gather_21_batch_dims_0 = const()[name = tensor("gather_21_batch_dims_0"), val = tensor(0)]; + tensor gather_21 = gather(axis = gather_21_axis_0, batch_dims = gather_21_batch_dims_0, indices = gather_21_indices_0, x = var_1723_shape)[name = tensor("gather_21")]; + tensor concat_25_values0_0 = const()[name = tensor("concat_25_values0_0"), val = tensor(0)]; + tensor concat_25_values1_0 = const()[name = tensor("concat_25_values1_0"), val = tensor(0)]; + tensor concat_25_values2_0 = const()[name = tensor("concat_25_values2_0"), val = tensor(0)]; + tensor concat_25_axis_0 = const()[name = tensor("concat_25_axis_0"), val = tensor(0)]; + tensor concat_25_interleave_0 = const()[name = tensor("concat_25_interleave_0"), val = tensor(false)]; + tensor concat_25 = concat(axis = concat_25_axis_0, interleave = concat_25_interleave_0, values = (concat_25_values0_0, concat_25_values1_0, concat_25_values2_0, gather_21))[name = tensor("concat_25")]; + tensor attention_mask_37_begin_0 = const()[name = tensor("attention_mask_37_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_37_end_mask_0 = const()[name = tensor("attention_mask_37_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_37 = slice_by_index(begin = attention_mask_37_begin_0, end = concat_25, end_mask = attention_mask_37_end_mask_0, x = reshape_4)[name = tensor("attention_mask_37")]; + tensor query_33 = transpose(perm = query_33_perm_0, x = var_1705)[name = tensor("transpose_21")]; + tensor mul_16 = mul(x = query_33, y = var_687)[name = tensor("mul_16")]; + tensor matmul_16_transpose_y_0 = const()[name = tensor("matmul_16_transpose_y_0"), val = tensor(true)]; + tensor matmul_16_transpose_x_0 = const()[name = tensor("matmul_16_transpose_x_0"), val = tensor(false)]; + tensor matmul_16 = matmul(transpose_x = matmul_16_transpose_x_0, transpose_y = matmul_16_transpose_y_0, x = mul_16, y = key_33)[name = tensor("matmul_16")]; + tensor add_16 = add(x = matmul_16, y = attention_mask_37)[name = tensor("add_16")]; + tensor softmax_16_axis_0 = const()[name = tensor("softmax_16_axis_0"), val = tensor(-1)]; + tensor softmax_16 = softmax(axis = softmax_16_axis_0, x = add_16)[name = tensor("softmax_16")]; + tensor attn_output_65_transpose_x_0 = const()[name = tensor("attn_output_65_transpose_x_0"), val = tensor(false)]; + tensor attn_output_65_transpose_y_0 = const()[name = tensor("attn_output_65_transpose_y_0"), val = tensor(false)]; + tensor attn_output_65 = matmul(transpose_x = attn_output_65_transpose_x_0, transpose_y = attn_output_65_transpose_y_0, x = softmax_16, y = value_33)[name = tensor("attn_output_65")]; + tensor var_1729_perm_0 = const()[name = tensor("op_1729_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1731 = const()[name = tensor("op_1731"), val = tensor([1, 1, -1])]; + tensor var_1729 = transpose(perm = var_1729_perm_0, x = attn_output_65)[name = tensor("transpose_20")]; + tensor var_1732 = reshape(shape = var_1731, x = var_1729)[name = tensor("op_1732")]; + tensor input_201 = linear(bias = decoder_layers_8_self_attn_out_proj_bias, weight = decoder_layers_8_self_attn_out_proj_weight_palettized, x = var_1732)[name = tensor("linear_67")]; + tensor input_203 = add(x = input_197, y = input_201)[name = tensor("input_203")]; + tensor hidden_states_85_axes_0 = const()[name = tensor("hidden_states_85_axes_0"), val = tensor([-1])]; + tensor hidden_states_85 = layer_norm(axes = hidden_states_85_axes_0, beta = decoder_layers_8_encoder_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_8_encoder_attn_layer_norm_weight, x = input_203)[name = tensor("hidden_states_85")]; + tensor var_1753 = linear(bias = decoder_layers_8_encoder_attn_q_proj_bias, weight = decoder_layers_8_encoder_attn_q_proj_weight_palettized, x = hidden_states_85)[name = tensor("linear_68")]; + tensor var_1754 = const()[name = tensor("op_1754"), val = tensor([1, 1, -1, 64])]; + tensor var_1755 = reshape(shape = var_1754, x = var_1753)[name = tensor("op_1755")]; + tensor query_35_perm_0 = const()[name = tensor("query_35_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1757_shape = shape(x = key_35)[name = tensor("op_1757_shape")]; + tensor gather_22_indices_0 = const()[name = tensor("gather_22_indices_0"), val = tensor(2)]; + tensor gather_22_axis_0 = const()[name = tensor("gather_22_axis_0"), val = tensor(0)]; + tensor gather_22_batch_dims_0 = const()[name = tensor("gather_22_batch_dims_0"), val = tensor(0)]; + tensor gather_22 = gather(axis = gather_22_axis_0, batch_dims = gather_22_batch_dims_0, indices = gather_22_indices_0, x = var_1757_shape)[name = tensor("gather_22")]; + tensor concat_26_values0_0 = const()[name = tensor("concat_26_values0_0"), val = tensor(0)]; + tensor concat_26_values1_0 = const()[name = tensor("concat_26_values1_0"), val = tensor(0)]; + tensor concat_26_values2_0 = const()[name = tensor("concat_26_values2_0"), val = tensor(0)]; + tensor concat_26_axis_0 = const()[name = tensor("concat_26_axis_0"), val = tensor(0)]; + tensor concat_26_interleave_0 = const()[name = tensor("concat_26_interleave_0"), val = tensor(false)]; + tensor concat_26 = concat(axis = concat_26_axis_0, interleave = concat_26_interleave_0, values = (concat_26_values0_0, concat_26_values1_0, concat_26_values2_0, gather_22))[name = tensor("concat_26")]; + tensor attention_mask_39_begin_0 = const()[name = tensor("attention_mask_39_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_39_end_mask_0 = const()[name = tensor("attention_mask_39_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_39 = slice_by_index(begin = attention_mask_39_begin_0, end = concat_26, end_mask = attention_mask_39_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_39")]; + tensor query_35 = transpose(perm = query_35_perm_0, x = var_1755)[name = tensor("transpose_19")]; + tensor mul_17 = mul(x = query_35, y = var_687)[name = tensor("mul_17")]; + tensor matmul_17_transpose_y_0 = const()[name = tensor("matmul_17_transpose_y_0"), val = tensor(true)]; + tensor matmul_17_transpose_x_0 = const()[name = tensor("matmul_17_transpose_x_0"), val = tensor(false)]; + tensor matmul_17 = matmul(transpose_x = matmul_17_transpose_x_0, transpose_y = matmul_17_transpose_y_0, x = mul_17, y = key_35)[name = tensor("matmul_17")]; + tensor add_17 = add(x = matmul_17, y = attention_mask_39)[name = tensor("add_17")]; + tensor softmax_17_axis_0 = const()[name = tensor("softmax_17_axis_0"), val = tensor(-1)]; + tensor softmax_17 = softmax(axis = softmax_17_axis_0, x = add_17)[name = tensor("softmax_17")]; + tensor attn_output_69_transpose_x_0 = const()[name = tensor("attn_output_69_transpose_x_0"), val = tensor(false)]; + tensor attn_output_69_transpose_y_0 = const()[name = tensor("attn_output_69_transpose_y_0"), val = tensor(false)]; + tensor attn_output_69 = matmul(transpose_x = attn_output_69_transpose_x_0, transpose_y = attn_output_69_transpose_y_0, x = softmax_17, y = value_35)[name = tensor("attn_output_69")]; + tensor var_1763_perm_0 = const()[name = tensor("op_1763_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1765 = const()[name = tensor("op_1765"), val = tensor([1, 1, -1])]; + tensor var_1763 = transpose(perm = var_1763_perm_0, x = attn_output_69)[name = tensor("transpose_18")]; + tensor var_1766 = reshape(shape = var_1765, x = var_1763)[name = tensor("op_1766")]; + tensor input_207 = linear(bias = decoder_layers_8_encoder_attn_out_proj_bias, weight = decoder_layers_8_encoder_attn_out_proj_weight_palettized, x = var_1766)[name = tensor("linear_69")]; + tensor input_209 = add(x = input_203, y = input_207)[name = tensor("input_209")]; + tensor input_211_axes_0 = const()[name = tensor("input_211_axes_0"), val = tensor([-1])]; + tensor input_211 = layer_norm(axes = input_211_axes_0, beta = decoder_layers_8_final_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_8_final_layer_norm_weight, x = input_209)[name = tensor("input_211")]; + tensor input_213 = linear(bias = decoder_layers_8_fc1_bias, weight = decoder_layers_8_fc1_weight_palettized, x = input_211)[name = tensor("linear_70")]; + tensor input_215 = relu(x = input_213)[name = tensor("input_215")]; + tensor input_219 = linear(bias = decoder_layers_8_fc2_bias, weight = decoder_layers_8_fc2_weight_palettized, x = input_215)[name = tensor("linear_71")]; + tensor input_221 = add(x = input_209, y = input_219)[name = tensor("input_221")]; + tensor hidden_states_91_axes_0 = const()[name = tensor("hidden_states_91_axes_0"), val = tensor([-1])]; + tensor hidden_states_91 = layer_norm(axes = hidden_states_91_axes_0, beta = decoder_layers_9_self_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_9_self_attn_layer_norm_weight, x = input_221)[name = tensor("hidden_states_91")]; + tensor var_1810 = linear(bias = decoder_layers_9_self_attn_q_proj_bias, weight = decoder_layers_9_self_attn_q_proj_weight_palettized, x = hidden_states_91)[name = tensor("linear_72")]; + tensor var_1811 = const()[name = tensor("op_1811"), val = tensor([1, 1, -1, 64])]; + tensor var_1812 = reshape(shape = var_1811, x = var_1810)[name = tensor("op_1812")]; + tensor query_37_perm_0 = const()[name = tensor("query_37_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_85 = linear(bias = decoder_layers_9_self_attn_k_proj_bias, weight = decoder_layers_9_self_attn_k_proj_weight_palettized, x = hidden_states_91)[name = tensor("linear_73")]; + tensor value_states_85 = linear(bias = decoder_layers_9_self_attn_v_proj_bias, weight = decoder_layers_9_self_attn_v_proj_weight_palettized, x = hidden_states_91)[name = tensor("linear_74")]; + tensor var_1820 = const()[name = tensor("op_1820"), val = tensor([1, 1, -1, 64])]; + tensor var_1821 = reshape(shape = var_1820, x = key_states_85)[name = tensor("op_1821")]; + tensor key_states_87_perm_0 = const()[name = tensor("key_states_87_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1823 = const()[name = tensor("op_1823"), val = tensor([1, 1, -1, 64])]; + tensor var_1824 = reshape(shape = var_1823, x = value_states_85)[name = tensor("op_1824")]; + tensor value_states_87_perm_0 = const()[name = tensor("value_states_87_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_37_interleave_0 = const()[name = tensor("key_37_interleave_0"), val = tensor(false)]; + tensor key_states_87 = transpose(perm = key_states_87_perm_0, x = var_1821)[name = tensor("transpose_17")]; + tensor key_37 = concat(axis = var_689, interleave = key_37_interleave_0, values = (var_570, key_states_87))[name = tensor("key_37")]; + tensor value_37_interleave_0 = const()[name = tensor("value_37_interleave_0"), val = tensor(false)]; + tensor value_states_87 = transpose(perm = value_states_87_perm_0, x = var_1824)[name = tensor("transpose_16")]; + tensor value_37 = concat(axis = var_689, interleave = value_37_interleave_0, values = (var_573, value_states_87))[name = tensor("value_37")]; + tensor var_1830_shape = shape(x = key_37)[name = tensor("op_1830_shape")]; + tensor gather_23_indices_0 = const()[name = tensor("gather_23_indices_0"), val = tensor(2)]; + tensor gather_23_axis_0 = const()[name = tensor("gather_23_axis_0"), val = tensor(0)]; + tensor gather_23_batch_dims_0 = const()[name = tensor("gather_23_batch_dims_0"), val = tensor(0)]; + tensor gather_23 = gather(axis = gather_23_axis_0, batch_dims = gather_23_batch_dims_0, indices = gather_23_indices_0, x = var_1830_shape)[name = tensor("gather_23")]; + tensor concat_27_values0_0 = const()[name = tensor("concat_27_values0_0"), val = tensor(0)]; + tensor concat_27_values1_0 = const()[name = tensor("concat_27_values1_0"), val = tensor(0)]; + tensor concat_27_values2_0 = const()[name = tensor("concat_27_values2_0"), val = tensor(0)]; + tensor concat_27_axis_0 = const()[name = tensor("concat_27_axis_0"), val = tensor(0)]; + tensor concat_27_interleave_0 = const()[name = tensor("concat_27_interleave_0"), val = tensor(false)]; + tensor concat_27 = concat(axis = concat_27_axis_0, interleave = concat_27_interleave_0, values = (concat_27_values0_0, concat_27_values1_0, concat_27_values2_0, gather_23))[name = tensor("concat_27")]; + tensor attention_mask_41_begin_0 = const()[name = tensor("attention_mask_41_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_41_end_mask_0 = const()[name = tensor("attention_mask_41_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_41 = slice_by_index(begin = attention_mask_41_begin_0, end = concat_27, end_mask = attention_mask_41_end_mask_0, x = reshape_4)[name = tensor("attention_mask_41")]; + tensor query_37 = transpose(perm = query_37_perm_0, x = var_1812)[name = tensor("transpose_15")]; + tensor mul_18 = mul(x = query_37, y = var_687)[name = tensor("mul_18")]; + tensor matmul_18_transpose_y_0 = const()[name = tensor("matmul_18_transpose_y_0"), val = tensor(true)]; + tensor matmul_18_transpose_x_0 = const()[name = tensor("matmul_18_transpose_x_0"), val = tensor(false)]; + tensor matmul_18 = matmul(transpose_x = matmul_18_transpose_x_0, transpose_y = matmul_18_transpose_y_0, x = mul_18, y = key_37)[name = tensor("matmul_18")]; + tensor add_18 = add(x = matmul_18, y = attention_mask_41)[name = tensor("add_18")]; + tensor softmax_18_axis_0 = const()[name = tensor("softmax_18_axis_0"), val = tensor(-1)]; + tensor softmax_18 = softmax(axis = softmax_18_axis_0, x = add_18)[name = tensor("softmax_18")]; + tensor attn_output_73_transpose_x_0 = const()[name = tensor("attn_output_73_transpose_x_0"), val = tensor(false)]; + tensor attn_output_73_transpose_y_0 = const()[name = tensor("attn_output_73_transpose_y_0"), val = tensor(false)]; + tensor attn_output_73 = matmul(transpose_x = attn_output_73_transpose_x_0, transpose_y = attn_output_73_transpose_y_0, x = softmax_18, y = value_37)[name = tensor("attn_output_73")]; + tensor var_1836_perm_0 = const()[name = tensor("op_1836_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1838 = const()[name = tensor("op_1838"), val = tensor([1, 1, -1])]; + tensor var_1836 = transpose(perm = var_1836_perm_0, x = attn_output_73)[name = tensor("transpose_14")]; + tensor var_1839 = reshape(shape = var_1838, x = var_1836)[name = tensor("op_1839")]; + tensor input_225 = linear(bias = decoder_layers_9_self_attn_out_proj_bias, weight = decoder_layers_9_self_attn_out_proj_weight_palettized, x = var_1839)[name = tensor("linear_75")]; + tensor input_227 = add(x = input_221, y = input_225)[name = tensor("input_227")]; + tensor hidden_states_95_axes_0 = const()[name = tensor("hidden_states_95_axes_0"), val = tensor([-1])]; + tensor hidden_states_95 = layer_norm(axes = hidden_states_95_axes_0, beta = decoder_layers_9_encoder_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_9_encoder_attn_layer_norm_weight, x = input_227)[name = tensor("hidden_states_95")]; + tensor var_1860 = linear(bias = decoder_layers_9_encoder_attn_q_proj_bias, weight = decoder_layers_9_encoder_attn_q_proj_weight_palettized, x = hidden_states_95)[name = tensor("linear_76")]; + tensor var_1861 = const()[name = tensor("op_1861"), val = tensor([1, 1, -1, 64])]; + tensor var_1862 = reshape(shape = var_1861, x = var_1860)[name = tensor("op_1862")]; + tensor query_39_perm_0 = const()[name = tensor("query_39_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1864_shape = shape(x = key_39)[name = tensor("op_1864_shape")]; + tensor gather_24_indices_0 = const()[name = tensor("gather_24_indices_0"), val = tensor(2)]; + tensor gather_24_axis_0 = const()[name = tensor("gather_24_axis_0"), val = tensor(0)]; + tensor gather_24_batch_dims_0 = const()[name = tensor("gather_24_batch_dims_0"), val = tensor(0)]; + tensor gather_24 = gather(axis = gather_24_axis_0, batch_dims = gather_24_batch_dims_0, indices = gather_24_indices_0, x = var_1864_shape)[name = tensor("gather_24")]; + tensor concat_28_values0_0 = const()[name = tensor("concat_28_values0_0"), val = tensor(0)]; + tensor concat_28_values1_0 = const()[name = tensor("concat_28_values1_0"), val = tensor(0)]; + tensor concat_28_values2_0 = const()[name = tensor("concat_28_values2_0"), val = tensor(0)]; + tensor concat_28_axis_0 = const()[name = tensor("concat_28_axis_0"), val = tensor(0)]; + tensor concat_28_interleave_0 = const()[name = tensor("concat_28_interleave_0"), val = tensor(false)]; + tensor concat_28 = concat(axis = concat_28_axis_0, interleave = concat_28_interleave_0, values = (concat_28_values0_0, concat_28_values1_0, concat_28_values2_0, gather_24))[name = tensor("concat_28")]; + tensor attention_mask_43_begin_0 = const()[name = tensor("attention_mask_43_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_43_end_mask_0 = const()[name = tensor("attention_mask_43_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_43 = slice_by_index(begin = attention_mask_43_begin_0, end = concat_28, end_mask = attention_mask_43_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_43")]; + tensor query_39 = transpose(perm = query_39_perm_0, x = var_1862)[name = tensor("transpose_13")]; + tensor mul_19 = mul(x = query_39, y = var_687)[name = tensor("mul_19")]; + tensor matmul_19_transpose_y_0 = const()[name = tensor("matmul_19_transpose_y_0"), val = tensor(true)]; + tensor matmul_19_transpose_x_0 = const()[name = tensor("matmul_19_transpose_x_0"), val = tensor(false)]; + tensor matmul_19 = matmul(transpose_x = matmul_19_transpose_x_0, transpose_y = matmul_19_transpose_y_0, x = mul_19, y = key_39)[name = tensor("matmul_19")]; + tensor add_19 = add(x = matmul_19, y = attention_mask_43)[name = tensor("add_19")]; + tensor softmax_19_axis_0 = const()[name = tensor("softmax_19_axis_0"), val = tensor(-1)]; + tensor softmax_19 = softmax(axis = softmax_19_axis_0, x = add_19)[name = tensor("softmax_19")]; + tensor attn_output_77_transpose_x_0 = const()[name = tensor("attn_output_77_transpose_x_0"), val = tensor(false)]; + tensor attn_output_77_transpose_y_0 = const()[name = tensor("attn_output_77_transpose_y_0"), val = tensor(false)]; + tensor attn_output_77 = matmul(transpose_x = attn_output_77_transpose_x_0, transpose_y = attn_output_77_transpose_y_0, x = softmax_19, y = value_39)[name = tensor("attn_output_77")]; + tensor var_1870_perm_0 = const()[name = tensor("op_1870_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1872 = const()[name = tensor("op_1872"), val = tensor([1, 1, -1])]; + tensor var_1870 = transpose(perm = var_1870_perm_0, x = attn_output_77)[name = tensor("transpose_12")]; + tensor var_1873 = reshape(shape = var_1872, x = var_1870)[name = tensor("op_1873")]; + tensor input_231 = linear(bias = decoder_layers_9_encoder_attn_out_proj_bias, weight = decoder_layers_9_encoder_attn_out_proj_weight_palettized, x = var_1873)[name = tensor("linear_77")]; + tensor input_233 = add(x = input_227, y = input_231)[name = tensor("input_233")]; + tensor input_235_axes_0 = const()[name = tensor("input_235_axes_0"), val = tensor([-1])]; + tensor input_235 = layer_norm(axes = input_235_axes_0, beta = decoder_layers_9_final_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_9_final_layer_norm_weight, x = input_233)[name = tensor("input_235")]; + tensor input_237 = linear(bias = decoder_layers_9_fc1_bias, weight = decoder_layers_9_fc1_weight_palettized, x = input_235)[name = tensor("linear_78")]; + tensor input_239 = relu(x = input_237)[name = tensor("input_239")]; + tensor input_243 = linear(bias = decoder_layers_9_fc2_bias, weight = decoder_layers_9_fc2_weight_palettized, x = input_239)[name = tensor("linear_79")]; + tensor input_245 = add(x = input_233, y = input_243)[name = tensor("input_245")]; + tensor hidden_states_101_axes_0 = const()[name = tensor("hidden_states_101_axes_0"), val = tensor([-1])]; + tensor hidden_states_101 = layer_norm(axes = hidden_states_101_axes_0, beta = decoder_layers_10_self_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_10_self_attn_layer_norm_weight, x = input_245)[name = tensor("hidden_states_101")]; + tensor var_1917 = linear(bias = decoder_layers_10_self_attn_q_proj_bias, weight = decoder_layers_10_self_attn_q_proj_weight_palettized, x = hidden_states_101)[name = tensor("linear_80")]; + tensor var_1918 = const()[name = tensor("op_1918"), val = tensor([1, 1, -1, 64])]; + tensor var_1919 = reshape(shape = var_1918, x = var_1917)[name = tensor("op_1919")]; + tensor query_41_perm_0 = const()[name = tensor("query_41_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_89 = linear(bias = decoder_layers_10_self_attn_k_proj_bias, weight = decoder_layers_10_self_attn_k_proj_weight_palettized, x = hidden_states_101)[name = tensor("linear_81")]; + tensor value_states_89 = linear(bias = decoder_layers_10_self_attn_v_proj_bias, weight = decoder_layers_10_self_attn_v_proj_weight_palettized, x = hidden_states_101)[name = tensor("linear_82")]; + tensor var_1927 = const()[name = tensor("op_1927"), val = tensor([1, 1, -1, 64])]; + tensor var_1928 = reshape(shape = var_1927, x = key_states_89)[name = tensor("op_1928")]; + tensor key_states_91_perm_0 = const()[name = tensor("key_states_91_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1930 = const()[name = tensor("op_1930"), val = tensor([1, 1, -1, 64])]; + tensor var_1931 = reshape(shape = var_1930, x = value_states_89)[name = tensor("op_1931")]; + tensor value_states_91_perm_0 = const()[name = tensor("value_states_91_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_41_interleave_0 = const()[name = tensor("key_41_interleave_0"), val = tensor(false)]; + tensor key_states_91 = transpose(perm = key_states_91_perm_0, x = var_1928)[name = tensor("transpose_11")]; + tensor key_41 = concat(axis = var_689, interleave = key_41_interleave_0, values = (var_614, key_states_91))[name = tensor("key_41")]; + tensor value_41_interleave_0 = const()[name = tensor("value_41_interleave_0"), val = tensor(false)]; + tensor value_states_91 = transpose(perm = value_states_91_perm_0, x = var_1931)[name = tensor("transpose_10")]; + tensor value_41 = concat(axis = var_689, interleave = value_41_interleave_0, values = (var_617, value_states_91))[name = tensor("value_41")]; + tensor var_1937_shape = shape(x = key_41)[name = tensor("op_1937_shape")]; + tensor gather_25_indices_0 = const()[name = tensor("gather_25_indices_0"), val = tensor(2)]; + tensor gather_25_axis_0 = const()[name = tensor("gather_25_axis_0"), val = tensor(0)]; + tensor gather_25_batch_dims_0 = const()[name = tensor("gather_25_batch_dims_0"), val = tensor(0)]; + tensor gather_25 = gather(axis = gather_25_axis_0, batch_dims = gather_25_batch_dims_0, indices = gather_25_indices_0, x = var_1937_shape)[name = tensor("gather_25")]; + tensor concat_29_values0_0 = const()[name = tensor("concat_29_values0_0"), val = tensor(0)]; + tensor concat_29_values1_0 = const()[name = tensor("concat_29_values1_0"), val = tensor(0)]; + tensor concat_29_values2_0 = const()[name = tensor("concat_29_values2_0"), val = tensor(0)]; + tensor concat_29_axis_0 = const()[name = tensor("concat_29_axis_0"), val = tensor(0)]; + tensor concat_29_interleave_0 = const()[name = tensor("concat_29_interleave_0"), val = tensor(false)]; + tensor concat_29 = concat(axis = concat_29_axis_0, interleave = concat_29_interleave_0, values = (concat_29_values0_0, concat_29_values1_0, concat_29_values2_0, gather_25))[name = tensor("concat_29")]; + tensor attention_mask_45_begin_0 = const()[name = tensor("attention_mask_45_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_45_end_mask_0 = const()[name = tensor("attention_mask_45_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_45 = slice_by_index(begin = attention_mask_45_begin_0, end = concat_29, end_mask = attention_mask_45_end_mask_0, x = reshape_4)[name = tensor("attention_mask_45")]; + tensor query_41 = transpose(perm = query_41_perm_0, x = var_1919)[name = tensor("transpose_9")]; + tensor mul_20 = mul(x = query_41, y = var_687)[name = tensor("mul_20")]; + tensor matmul_20_transpose_y_0 = const()[name = tensor("matmul_20_transpose_y_0"), val = tensor(true)]; + tensor matmul_20_transpose_x_0 = const()[name = tensor("matmul_20_transpose_x_0"), val = tensor(false)]; + tensor matmul_20 = matmul(transpose_x = matmul_20_transpose_x_0, transpose_y = matmul_20_transpose_y_0, x = mul_20, y = key_41)[name = tensor("matmul_20")]; + tensor add_20 = add(x = matmul_20, y = attention_mask_45)[name = tensor("add_20")]; + tensor softmax_20_axis_0 = const()[name = tensor("softmax_20_axis_0"), val = tensor(-1)]; + tensor softmax_20 = softmax(axis = softmax_20_axis_0, x = add_20)[name = tensor("softmax_20")]; + tensor attn_output_81_transpose_x_0 = const()[name = tensor("attn_output_81_transpose_x_0"), val = tensor(false)]; + tensor attn_output_81_transpose_y_0 = const()[name = tensor("attn_output_81_transpose_y_0"), val = tensor(false)]; + tensor attn_output_81 = matmul(transpose_x = attn_output_81_transpose_x_0, transpose_y = attn_output_81_transpose_y_0, x = softmax_20, y = value_41)[name = tensor("attn_output_81")]; + tensor var_1943_perm_0 = const()[name = tensor("op_1943_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1945 = const()[name = tensor("op_1945"), val = tensor([1, 1, -1])]; + tensor var_1943 = transpose(perm = var_1943_perm_0, x = attn_output_81)[name = tensor("transpose_8")]; + tensor var_1946 = reshape(shape = var_1945, x = var_1943)[name = tensor("op_1946")]; + tensor input_249 = linear(bias = decoder_layers_10_self_attn_out_proj_bias, weight = decoder_layers_10_self_attn_out_proj_weight_palettized, x = var_1946)[name = tensor("linear_83")]; + tensor input_251 = add(x = input_245, y = input_249)[name = tensor("input_251")]; + tensor hidden_states_105_axes_0 = const()[name = tensor("hidden_states_105_axes_0"), val = tensor([-1])]; + tensor hidden_states_105 = layer_norm(axes = hidden_states_105_axes_0, beta = decoder_layers_10_encoder_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_10_encoder_attn_layer_norm_weight, x = input_251)[name = tensor("hidden_states_105")]; + tensor var_1967 = linear(bias = decoder_layers_10_encoder_attn_q_proj_bias, weight = decoder_layers_10_encoder_attn_q_proj_weight_palettized, x = hidden_states_105)[name = tensor("linear_84")]; + tensor var_1968 = const()[name = tensor("op_1968"), val = tensor([1, 1, -1, 64])]; + tensor var_1969 = reshape(shape = var_1968, x = var_1967)[name = tensor("op_1969")]; + tensor query_43_perm_0 = const()[name = tensor("query_43_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1971_shape = shape(x = key_43)[name = tensor("op_1971_shape")]; + tensor gather_26_indices_0 = const()[name = tensor("gather_26_indices_0"), val = tensor(2)]; + tensor gather_26_axis_0 = const()[name = tensor("gather_26_axis_0"), val = tensor(0)]; + tensor gather_26_batch_dims_0 = const()[name = tensor("gather_26_batch_dims_0"), val = tensor(0)]; + tensor gather_26 = gather(axis = gather_26_axis_0, batch_dims = gather_26_batch_dims_0, indices = gather_26_indices_0, x = var_1971_shape)[name = tensor("gather_26")]; + tensor concat_30_values0_0 = const()[name = tensor("concat_30_values0_0"), val = tensor(0)]; + tensor concat_30_values1_0 = const()[name = tensor("concat_30_values1_0"), val = tensor(0)]; + tensor concat_30_values2_0 = const()[name = tensor("concat_30_values2_0"), val = tensor(0)]; + tensor concat_30_axis_0 = const()[name = tensor("concat_30_axis_0"), val = tensor(0)]; + tensor concat_30_interleave_0 = const()[name = tensor("concat_30_interleave_0"), val = tensor(false)]; + tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (concat_30_values0_0, concat_30_values1_0, concat_30_values2_0, gather_26))[name = tensor("concat_30")]; + tensor attention_mask_47_begin_0 = const()[name = tensor("attention_mask_47_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_47_end_mask_0 = const()[name = tensor("attention_mask_47_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_47 = slice_by_index(begin = attention_mask_47_begin_0, end = concat_30, end_mask = attention_mask_47_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_47")]; + tensor query_43 = transpose(perm = query_43_perm_0, x = var_1969)[name = tensor("transpose_7")]; + tensor mul_21 = mul(x = query_43, y = var_687)[name = tensor("mul_21")]; + tensor matmul_21_transpose_y_0 = const()[name = tensor("matmul_21_transpose_y_0"), val = tensor(true)]; + tensor matmul_21_transpose_x_0 = const()[name = tensor("matmul_21_transpose_x_0"), val = tensor(false)]; + tensor matmul_21 = matmul(transpose_x = matmul_21_transpose_x_0, transpose_y = matmul_21_transpose_y_0, x = mul_21, y = key_43)[name = tensor("matmul_21")]; + tensor add_21 = add(x = matmul_21, y = attention_mask_47)[name = tensor("add_21")]; + tensor softmax_21_axis_0 = const()[name = tensor("softmax_21_axis_0"), val = tensor(-1)]; + tensor softmax_21 = softmax(axis = softmax_21_axis_0, x = add_21)[name = tensor("softmax_21")]; + tensor attn_output_85_transpose_x_0 = const()[name = tensor("attn_output_85_transpose_x_0"), val = tensor(false)]; + tensor attn_output_85_transpose_y_0 = const()[name = tensor("attn_output_85_transpose_y_0"), val = tensor(false)]; + tensor attn_output_85 = matmul(transpose_x = attn_output_85_transpose_x_0, transpose_y = attn_output_85_transpose_y_0, x = softmax_21, y = value_43)[name = tensor("attn_output_85")]; + tensor var_1977_perm_0 = const()[name = tensor("op_1977_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1979 = const()[name = tensor("op_1979"), val = tensor([1, 1, -1])]; + tensor var_1977 = transpose(perm = var_1977_perm_0, x = attn_output_85)[name = tensor("transpose_6")]; + tensor var_1980 = reshape(shape = var_1979, x = var_1977)[name = tensor("op_1980")]; + tensor input_255 = linear(bias = decoder_layers_10_encoder_attn_out_proj_bias, weight = decoder_layers_10_encoder_attn_out_proj_weight_palettized, x = var_1980)[name = tensor("linear_85")]; + tensor input_257 = add(x = input_251, y = input_255)[name = tensor("input_257")]; + tensor input_259_axes_0 = const()[name = tensor("input_259_axes_0"), val = tensor([-1])]; + tensor input_259 = layer_norm(axes = input_259_axes_0, beta = decoder_layers_10_final_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_10_final_layer_norm_weight, x = input_257)[name = tensor("input_259")]; + tensor input_261 = linear(bias = decoder_layers_10_fc1_bias, weight = decoder_layers_10_fc1_weight_palettized, x = input_259)[name = tensor("linear_86")]; + tensor input_263 = relu(x = input_261)[name = tensor("input_263")]; + tensor input_267 = linear(bias = decoder_layers_10_fc2_bias, weight = decoder_layers_10_fc2_weight_palettized, x = input_263)[name = tensor("linear_87")]; + tensor input_269 = add(x = input_257, y = input_267)[name = tensor("input_269")]; + tensor hidden_states_111_axes_0 = const()[name = tensor("hidden_states_111_axes_0"), val = tensor([-1])]; + tensor hidden_states_111 = layer_norm(axes = hidden_states_111_axes_0, beta = decoder_layers_11_self_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_11_self_attn_layer_norm_weight, x = input_269)[name = tensor("hidden_states_111")]; + tensor var_2024 = linear(bias = decoder_layers_11_self_attn_q_proj_bias, weight = decoder_layers_11_self_attn_q_proj_weight_palettized, x = hidden_states_111)[name = tensor("linear_88")]; + tensor var_2025 = const()[name = tensor("op_2025"), val = tensor([1, 1, -1, 64])]; + tensor var_2026 = reshape(shape = var_2025, x = var_2024)[name = tensor("op_2026")]; + tensor query_45_perm_0 = const()[name = tensor("query_45_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_93 = linear(bias = decoder_layers_11_self_attn_k_proj_bias, weight = decoder_layers_11_self_attn_k_proj_weight_palettized, x = hidden_states_111)[name = tensor("linear_89")]; + tensor value_states_93 = linear(bias = decoder_layers_11_self_attn_v_proj_bias, weight = decoder_layers_11_self_attn_v_proj_weight_palettized, x = hidden_states_111)[name = tensor("linear_90")]; + tensor var_2034 = const()[name = tensor("op_2034"), val = tensor([1, 1, -1, 64])]; + tensor var_2035 = reshape(shape = var_2034, x = key_states_93)[name = tensor("op_2035")]; + tensor key_states_perm_0 = const()[name = tensor("key_states_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_2037 = const()[name = tensor("op_2037"), val = tensor([1, 1, -1, 64])]; + tensor var_2038 = reshape(shape = var_2037, x = value_states_93)[name = tensor("op_2038")]; + tensor value_states_perm_0 = const()[name = tensor("value_states_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_45_interleave_0 = const()[name = tensor("key_45_interleave_0"), val = tensor(false)]; + tensor key_states = transpose(perm = key_states_perm_0, x = var_2035)[name = tensor("transpose_5")]; + tensor key_45 = concat(axis = var_689, interleave = key_45_interleave_0, values = (var_658, key_states))[name = tensor("key_45")]; + tensor value_45_interleave_0 = const()[name = tensor("value_45_interleave_0"), val = tensor(false)]; + tensor value_states = transpose(perm = value_states_perm_0, x = var_2038)[name = tensor("transpose_4")]; + tensor value_45 = concat(axis = var_689, interleave = value_45_interleave_0, values = (var_661, value_states))[name = tensor("value_45")]; + tensor var_2044_shape = shape(x = key_45)[name = tensor("op_2044_shape")]; + tensor gather_27_indices_0 = const()[name = tensor("gather_27_indices_0"), val = tensor(2)]; + tensor gather_27_axis_0 = const()[name = tensor("gather_27_axis_0"), val = tensor(0)]; + tensor gather_27_batch_dims_0 = const()[name = tensor("gather_27_batch_dims_0"), val = tensor(0)]; + tensor gather_27 = gather(axis = gather_27_axis_0, batch_dims = gather_27_batch_dims_0, indices = gather_27_indices_0, x = var_2044_shape)[name = tensor("gather_27")]; + tensor concat_31_values0_0 = const()[name = tensor("concat_31_values0_0"), val = tensor(0)]; + tensor concat_31_values1_0 = const()[name = tensor("concat_31_values1_0"), val = tensor(0)]; + tensor concat_31_values2_0 = const()[name = tensor("concat_31_values2_0"), val = tensor(0)]; + tensor concat_31_axis_0 = const()[name = tensor("concat_31_axis_0"), val = tensor(0)]; + tensor concat_31_interleave_0 = const()[name = tensor("concat_31_interleave_0"), val = tensor(false)]; + tensor concat_31 = concat(axis = concat_31_axis_0, interleave = concat_31_interleave_0, values = (concat_31_values0_0, concat_31_values1_0, concat_31_values2_0, gather_27))[name = tensor("concat_31")]; + tensor attention_mask_49_begin_0 = const()[name = tensor("attention_mask_49_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_49_end_mask_0 = const()[name = tensor("attention_mask_49_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_49 = slice_by_index(begin = attention_mask_49_begin_0, end = concat_31, end_mask = attention_mask_49_end_mask_0, x = reshape_4)[name = tensor("attention_mask_49")]; + tensor query_45 = transpose(perm = query_45_perm_0, x = var_2026)[name = tensor("transpose_3")]; + tensor mul_22 = mul(x = query_45, y = var_687)[name = tensor("mul_22")]; + tensor matmul_22_transpose_y_0 = const()[name = tensor("matmul_22_transpose_y_0"), val = tensor(true)]; + tensor matmul_22_transpose_x_0 = const()[name = tensor("matmul_22_transpose_x_0"), val = tensor(false)]; + tensor matmul_22 = matmul(transpose_x = matmul_22_transpose_x_0, transpose_y = matmul_22_transpose_y_0, x = mul_22, y = key_45)[name = tensor("matmul_22")]; + tensor add_22 = add(x = matmul_22, y = attention_mask_49)[name = tensor("add_22")]; + tensor softmax_22_axis_0 = const()[name = tensor("softmax_22_axis_0"), val = tensor(-1)]; + tensor softmax_22 = softmax(axis = softmax_22_axis_0, x = add_22)[name = tensor("softmax_22")]; + tensor attn_output_89_transpose_x_0 = const()[name = tensor("attn_output_89_transpose_x_0"), val = tensor(false)]; + tensor attn_output_89_transpose_y_0 = const()[name = tensor("attn_output_89_transpose_y_0"), val = tensor(false)]; + tensor attn_output_89 = matmul(transpose_x = attn_output_89_transpose_x_0, transpose_y = attn_output_89_transpose_y_0, x = softmax_22, y = value_45)[name = tensor("attn_output_89")]; + tensor var_2050_perm_0 = const()[name = tensor("op_2050_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_2052 = const()[name = tensor("op_2052"), val = tensor([1, 1, -1])]; + tensor var_2050 = transpose(perm = var_2050_perm_0, x = attn_output_89)[name = tensor("transpose_2")]; + tensor var_2053 = reshape(shape = var_2052, x = var_2050)[name = tensor("op_2053")]; + tensor input_273 = linear(bias = decoder_layers_11_self_attn_out_proj_bias, weight = decoder_layers_11_self_attn_out_proj_weight_palettized, x = var_2053)[name = tensor("linear_91")]; + tensor input_275 = add(x = input_269, y = input_273)[name = tensor("input_275")]; + tensor hidden_states_115_axes_0 = const()[name = tensor("hidden_states_115_axes_0"), val = tensor([-1])]; + tensor hidden_states_115 = layer_norm(axes = hidden_states_115_axes_0, beta = decoder_layers_11_encoder_attn_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_11_encoder_attn_layer_norm_weight, x = input_275)[name = tensor("hidden_states_115")]; + tensor var_2074 = linear(bias = decoder_layers_11_encoder_attn_q_proj_bias, weight = decoder_layers_11_encoder_attn_q_proj_weight_palettized, x = hidden_states_115)[name = tensor("linear_92")]; + tensor var_2075 = const()[name = tensor("op_2075"), val = tensor([1, 1, -1, 64])]; + tensor var_2076 = reshape(shape = var_2075, x = var_2074)[name = tensor("op_2076")]; + tensor query_perm_0 = const()[name = tensor("query_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_2078_shape = shape(x = key)[name = tensor("op_2078_shape")]; + tensor gather_28_indices_0 = const()[name = tensor("gather_28_indices_0"), val = tensor(2)]; + tensor gather_28_axis_0 = const()[name = tensor("gather_28_axis_0"), val = tensor(0)]; + tensor gather_28_batch_dims_0 = const()[name = tensor("gather_28_batch_dims_0"), val = tensor(0)]; + tensor gather_28 = gather(axis = gather_28_axis_0, batch_dims = gather_28_batch_dims_0, indices = gather_28_indices_0, x = var_2078_shape)[name = tensor("gather_28")]; + tensor concat_32_values0_0 = const()[name = tensor("concat_32_values0_0"), val = tensor(0)]; + tensor concat_32_values1_0 = const()[name = tensor("concat_32_values1_0"), val = tensor(0)]; + tensor concat_32_values2_0 = const()[name = tensor("concat_32_values2_0"), val = tensor(0)]; + tensor concat_32_axis_0 = const()[name = tensor("concat_32_axis_0"), val = tensor(0)]; + tensor concat_32_interleave_0 = const()[name = tensor("concat_32_interleave_0"), val = tensor(false)]; + tensor concat_32 = concat(axis = concat_32_axis_0, interleave = concat_32_interleave_0, values = (concat_32_values0_0, concat_32_values1_0, concat_32_values2_0, gather_28))[name = tensor("concat_32")]; + tensor attention_mask_begin_0 = const()[name = tensor("attention_mask_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_end_mask_0 = const()[name = tensor("attention_mask_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask = slice_by_index(begin = attention_mask_begin_0, end = concat_32, end_mask = attention_mask_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask")]; + tensor query = transpose(perm = query_perm_0, x = var_2076)[name = tensor("transpose_1")]; + tensor mul_23 = mul(x = query, y = var_687)[name = tensor("mul_23")]; + tensor matmul_23_transpose_y_0 = const()[name = tensor("matmul_23_transpose_y_0"), val = tensor(true)]; + tensor matmul_23_transpose_x_0 = const()[name = tensor("matmul_23_transpose_x_0"), val = tensor(false)]; + tensor matmul_23 = matmul(transpose_x = matmul_23_transpose_x_0, transpose_y = matmul_23_transpose_y_0, x = mul_23, y = key)[name = tensor("matmul_23")]; + tensor add_23 = add(x = matmul_23, y = attention_mask)[name = tensor("add_23")]; + tensor softmax_23_axis_0 = const()[name = tensor("softmax_23_axis_0"), val = tensor(-1)]; + tensor softmax_23 = softmax(axis = softmax_23_axis_0, x = add_23)[name = tensor("softmax_23")]; + tensor attn_output_93_transpose_x_0 = const()[name = tensor("attn_output_93_transpose_x_0"), val = tensor(false)]; + tensor attn_output_93_transpose_y_0 = const()[name = tensor("attn_output_93_transpose_y_0"), val = tensor(false)]; + tensor attn_output_93 = matmul(transpose_x = attn_output_93_transpose_x_0, transpose_y = attn_output_93_transpose_y_0, x = softmax_23, y = value)[name = tensor("attn_output_93")]; + tensor var_2084_perm_0 = const()[name = tensor("op_2084_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_2086 = const()[name = tensor("op_2086"), val = tensor([1, 1, -1])]; + tensor var_2084 = transpose(perm = var_2084_perm_0, x = attn_output_93)[name = tensor("transpose_0")]; + tensor var_2087 = reshape(shape = var_2086, x = var_2084)[name = tensor("op_2087")]; + tensor input_279 = linear(bias = decoder_layers_11_encoder_attn_out_proj_bias, weight = decoder_layers_11_encoder_attn_out_proj_weight_palettized, x = var_2087)[name = tensor("linear_93")]; + tensor input_281 = add(x = input_275, y = input_279)[name = tensor("input_281")]; + tensor input_283_axes_0 = const()[name = tensor("input_283_axes_0"), val = tensor([-1])]; + tensor input_283 = layer_norm(axes = input_283_axes_0, beta = decoder_layers_11_final_layer_norm_bias, epsilon = var_685, gamma = decoder_layers_11_final_layer_norm_weight, x = input_281)[name = tensor("input_283")]; + tensor input_285 = linear(bias = decoder_layers_11_fc1_bias, weight = decoder_layers_11_fc1_weight_palettized, x = input_283)[name = tensor("linear_94")]; + tensor input_287 = relu(x = input_285)[name = tensor("input_287")]; + tensor input_291 = linear(bias = decoder_layers_11_fc2_bias, weight = decoder_layers_11_fc2_weight_palettized, x = input_287)[name = tensor("linear_95")]; + tensor input_293 = add(x = input_281, y = input_291)[name = tensor("input_293")]; + tensor input_axes_0 = const()[name = tensor("input_axes_0"), val = tensor([-1])]; + tensor input = layer_norm(axes = input_axes_0, beta = decoder_layer_norm_bias, epsilon = var_685, gamma = decoder_layer_norm_weight, x = input_293)[name = tensor("input")]; + tensor linear_96_bias_0 = const()[name = tensor("linear_96_bias_0"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(440526336)))]; + tensor logits = linear(bias = linear_96_bias_0, weight = decoder_embed_tokens_weight_palettized, x = input)[name = tensor("linear_96")]; + tensor var_2146_axis_0 = const()[name = tensor("op_2146_axis_0"), val = tensor(0)]; + tensor new_past_self_key = stack(axis = var_2146_axis_0, values = (key_1, key_5, key_9, key_13, key_17, key_21, key_25, key_29, key_33, key_37, key_41, key_45))[name = tensor("op_2146")]; + tensor var_2149_axis_0 = const()[name = tensor("op_2149_axis_0"), val = tensor(0)]; + tensor new_past_self_value = stack(axis = var_2149_axis_0, values = (value_1, value_5, value_9, value_13, value_17, value_21, value_25, value_29, value_33, value_37, value_41, value_45))[name = tensor("op_2149")]; + tensor var_2152_axis_0 = const()[name = tensor("op_2152_axis_0"), val = tensor(0)]; + tensor new_past_cross_key = stack(axis = var_2152_axis_0, values = (key_3, key_7, key_11, key_15, key_19, key_23, key_27, key_31, key_35, key_39, key_43, key))[name = tensor("op_2152")]; + tensor var_2155_axis_0 = const()[name = tensor("op_2155_axis_0"), val = tensor(0)]; + tensor new_past_cross_value = stack(axis = var_2155_axis_0, values = (value_3, value_7, value_11, value_15, value_19, value_23, value_27, value_31, value_35, value_39, value_43, value))[name = tensor("op_2155")]; + tensor encoder_hidden_states_tmp = identity(x = encoder_hidden_states)[name = tensor("encoder_hidden_states_tmp")]; + } -> (logits, new_past_self_key, new_past_self_value, new_past_cross_key, new_past_cross_value); +} \ No newline at end of file