diff --git "a/NLLB_Decoder_1024_init.mlmodelc/model.mil" "b/NLLB_Decoder_1024_init.mlmodelc/model.mil" new file mode 100644--- /dev/null +++ "b/NLLB_Decoder_1024_init.mlmodelc/model.mil" @@ -0,0 +1,1564 @@ +program(1.0) +[buildInfo = dict, tensor>({{"coremlc-component-MIL", "3520.4.1"}, {"coremlc-version", "3520.5.1"}})] +{ + func main(tensor encoder_attention_mask, tensor encoder_hidden_states, tensor input_ids) [FlexibleShapeInformation = tuple, dict, tensor>>, tuple, dict, list, ?>>>>((("DefaultShapes", {{"encoder_attention_mask", [1, 1]}, {"encoder_hidden_states", [1, 1, 1024]}}), ("RangeDims", {{"encoder_attention_mask", [[1, 1], [1, 1024]]}, {"encoder_hidden_states", [[1, 1], [1, 1024], [1024, 1024]]}})))] { + tensor decoder_embed_tokens_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(64))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(262355072))), name = tensor("decoder_embed_tokens_weight_palettized"), shape = tensor([256206, 1024])]; + tensor decoder_embed_positions_weights_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(262356160))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263406848))), name = tensor("decoder_embed_positions_weights_palettized"), shape = tensor([1026, 1024])]; + tensor decoder_layers_0_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_0_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263407936)))]; + tensor decoder_layers_0_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_0_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263412096)))]; + tensor decoder_layers_0_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_0_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263416256)))]; + tensor decoder_layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263420416))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(264469056))), name = tensor("decoder_layers_0_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_0_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_0_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(264470144)))]; + tensor decoder_layers_0_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(264474304))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(265522944))), name = tensor("decoder_layers_0_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_0_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_0_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(265524032)))]; + tensor decoder_layers_0_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(265528192))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(266576832))), name = tensor("decoder_layers_0_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_0_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_0_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(266577920)))]; + tensor decoder_layers_0_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(266582080))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(267630720))), name = tensor("decoder_layers_0_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_0_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_0_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(267631808)))]; + tensor decoder_layers_0_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_0_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(267635968)))]; + tensor decoder_layers_0_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_0_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(267640128)))]; + tensor decoder_layers_0_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(267644288))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(268692928))), name = tensor("decoder_layers_0_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_0_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_0_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(268694016)))]; + tensor decoder_layers_0_encoder_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(268698176))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(269746816))), name = tensor("decoder_layers_0_encoder_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_0_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_0_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(269747904)))]; + tensor decoder_layers_0_encoder_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(269752064))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(270800704))), name = tensor("decoder_layers_0_encoder_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_0_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_0_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(270801792)))]; + tensor decoder_layers_0_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(270805952))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(271854592))), name = tensor("decoder_layers_0_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_0_final_layer_norm_bias = const()[name = tensor("decoder_layers_0_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(271855680)))]; + tensor decoder_layers_0_final_layer_norm_weight = const()[name = tensor("decoder_layers_0_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(271859840)))]; + tensor decoder_layers_0_fc1_bias = const()[name = tensor("decoder_layers_0_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(271864000)))]; + tensor decoder_layers_0_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(271880448))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(276074816))), name = tensor("decoder_layers_0_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_0_fc2_bias = const()[name = tensor("decoder_layers_0_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(276075904)))]; + tensor decoder_layers_0_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(276080064))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(280274432))), name = tensor("decoder_layers_0_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_1_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_1_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(280275520)))]; + tensor decoder_layers_1_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_1_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(280279680)))]; + tensor decoder_layers_1_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_1_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(280283840)))]; + tensor decoder_layers_1_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(280288000))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(281336640))), name = tensor("decoder_layers_1_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_1_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_1_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(281337728)))]; + tensor decoder_layers_1_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(281341888))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(282390528))), name = tensor("decoder_layers_1_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_1_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_1_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(282391616)))]; + tensor decoder_layers_1_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(282395776))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(283444416))), name = tensor("decoder_layers_1_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_1_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_1_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(283445504)))]; + tensor decoder_layers_1_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(283449664))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(284498304))), name = tensor("decoder_layers_1_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_1_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_1_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(284499392)))]; + tensor decoder_layers_1_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_1_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(284503552)))]; + tensor decoder_layers_1_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_1_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(284507712)))]; + tensor decoder_layers_1_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(284511872))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(285560512))), name = tensor("decoder_layers_1_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_1_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_1_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(285561600)))]; + tensor decoder_layers_1_encoder_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(285565760))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(286614400))), name = tensor("decoder_layers_1_encoder_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_1_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_1_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(286615488)))]; + tensor decoder_layers_1_encoder_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(286619648))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(287668288))), name = tensor("decoder_layers_1_encoder_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_1_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_1_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(287669376)))]; + tensor decoder_layers_1_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(287673536))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(284498304))), name = tensor("decoder_layers_1_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_1_final_layer_norm_bias = const()[name = tensor("decoder_layers_1_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(288722176)))]; + tensor decoder_layers_1_final_layer_norm_weight = const()[name = tensor("decoder_layers_1_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(288726336)))]; + tensor decoder_layers_1_fc1_bias = const()[name = tensor("decoder_layers_1_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(288730496)))]; + tensor decoder_layers_1_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(288746944))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(284498304))), name = tensor("decoder_layers_1_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_1_fc2_bias = const()[name = tensor("decoder_layers_1_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(292941312)))]; + tensor decoder_layers_1_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(292945472))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(297139840))), name = tensor("decoder_layers_1_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_2_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_2_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(297140928)))]; + tensor decoder_layers_2_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_2_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(297145088)))]; + tensor decoder_layers_2_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_2_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(297149248)))]; + tensor decoder_layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(297153408))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(298202048))), name = tensor("decoder_layers_2_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_2_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_2_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(298203136)))]; + tensor decoder_layers_2_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(298207296))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(299255936))), name = tensor("decoder_layers_2_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_2_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_2_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(299257024)))]; + tensor decoder_layers_2_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(299261184))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(300309824))), name = tensor("decoder_layers_2_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_2_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_2_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(300310912)))]; + tensor decoder_layers_2_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(300315072))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(301363712))), name = tensor("decoder_layers_2_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_2_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_2_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(301364800)))]; + tensor decoder_layers_2_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_2_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(301368960)))]; + tensor decoder_layers_2_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_2_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(301373120)))]; + tensor decoder_layers_2_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(301377280))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(302425920))), name = tensor("decoder_layers_2_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_2_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_2_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(302427008)))]; + tensor decoder_layers_2_encoder_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(302431168))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(303479808))), name = tensor("decoder_layers_2_encoder_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_2_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_2_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(303480896)))]; + tensor decoder_layers_2_encoder_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(303485056))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(304533696))), name = tensor("decoder_layers_2_encoder_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_2_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_2_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(304534784)))]; + tensor decoder_layers_2_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(304538944))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(305587584))), name = tensor("decoder_layers_2_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_2_final_layer_norm_bias = const()[name = tensor("decoder_layers_2_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(305588672)))]; + tensor decoder_layers_2_final_layer_norm_weight = const()[name = tensor("decoder_layers_2_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(305592832)))]; + tensor decoder_layers_2_fc1_bias = const()[name = tensor("decoder_layers_2_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(305596992)))]; + tensor decoder_layers_2_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(305613440))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(309807808))), name = tensor("decoder_layers_2_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_2_fc2_bias = const()[name = tensor("decoder_layers_2_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(309808896)))]; + tensor decoder_layers_2_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(309813056))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(314007424))), name = tensor("decoder_layers_2_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_3_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_3_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(314008512)))]; + tensor decoder_layers_3_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_3_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(314012672)))]; + tensor decoder_layers_3_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_3_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(314016832)))]; + tensor decoder_layers_3_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(314020992))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(315069632))), name = tensor("decoder_layers_3_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_3_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_3_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(315070720)))]; + tensor decoder_layers_3_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(315074880))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(316123520))), name = tensor("decoder_layers_3_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_3_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_3_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(316124608)))]; + tensor decoder_layers_3_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(316128768))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(317177408))), name = tensor("decoder_layers_3_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_3_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_3_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(317178496)))]; + tensor decoder_layers_3_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(317182656))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(318231296))), name = tensor("decoder_layers_3_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_3_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_3_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(318232384)))]; + tensor decoder_layers_3_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_3_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(318236544)))]; + tensor decoder_layers_3_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_3_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(318240704)))]; + tensor decoder_layers_3_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(318244864))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(319293504))), name = tensor("decoder_layers_3_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_3_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_3_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(319294592)))]; + tensor decoder_layers_3_encoder_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(319298752))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(320347392))), name = tensor("decoder_layers_3_encoder_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_3_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_3_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(320348480)))]; + tensor decoder_layers_3_encoder_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(320352640))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(321401280))), name = tensor("decoder_layers_3_encoder_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_3_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_3_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(321402368)))]; + tensor decoder_layers_3_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(321406528))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(322455168))), name = tensor("decoder_layers_3_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_3_final_layer_norm_bias = const()[name = tensor("decoder_layers_3_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(322456256)))]; + tensor decoder_layers_3_final_layer_norm_weight = const()[name = tensor("decoder_layers_3_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(322460416)))]; + tensor decoder_layers_3_fc1_bias = const()[name = tensor("decoder_layers_3_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(322464576)))]; + tensor decoder_layers_3_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(322481024))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(326675392))), name = tensor("decoder_layers_3_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_3_fc2_bias = const()[name = tensor("decoder_layers_3_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(326676480)))]; + tensor decoder_layers_3_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(326680640))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(330875008))), name = tensor("decoder_layers_3_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_4_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_4_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(330876096)))]; + tensor decoder_layers_4_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_4_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(330880256)))]; + tensor decoder_layers_4_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_4_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(330884416)))]; + tensor decoder_layers_4_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(330888576))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(331937216))), name = tensor("decoder_layers_4_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_4_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_4_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(331938304)))]; + tensor decoder_layers_4_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(331942464))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(332991104))), name = tensor("decoder_layers_4_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_4_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_4_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(332992192)))]; + tensor decoder_layers_4_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(332996352))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(334044992))), name = tensor("decoder_layers_4_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_4_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_4_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(334046080)))]; + tensor decoder_layers_4_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(334050240))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(335098880))), name = tensor("decoder_layers_4_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_4_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_4_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(335099968)))]; + tensor decoder_layers_4_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_4_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(335104128)))]; + tensor decoder_layers_4_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_4_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(335108288)))]; + tensor decoder_layers_4_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(335112448))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(336161088))), name = tensor("decoder_layers_4_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_4_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_4_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(336162176)))]; + tensor decoder_layers_4_encoder_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(336166336))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(337214976))), name = tensor("decoder_layers_4_encoder_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_4_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_4_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(337216064)))]; + tensor decoder_layers_4_encoder_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(337220224))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(338268864))), name = tensor("decoder_layers_4_encoder_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_4_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_4_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(338269952)))]; + tensor decoder_layers_4_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(338274112))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(339322752))), name = tensor("decoder_layers_4_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_4_final_layer_norm_bias = const()[name = tensor("decoder_layers_4_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(339323840)))]; + tensor decoder_layers_4_final_layer_norm_weight = const()[name = tensor("decoder_layers_4_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(339328000)))]; + tensor decoder_layers_4_fc1_bias = const()[name = tensor("decoder_layers_4_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(339332160)))]; + tensor decoder_layers_4_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(339348608))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(343542976))), name = tensor("decoder_layers_4_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_4_fc2_bias = const()[name = tensor("decoder_layers_4_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(343544064)))]; + tensor decoder_layers_4_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(343548224))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(347742592))), name = tensor("decoder_layers_4_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_5_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_5_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(347743680)))]; + tensor decoder_layers_5_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_5_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(347747840)))]; + tensor decoder_layers_5_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_5_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(347752000)))]; + tensor decoder_layers_5_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(347756160))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(348804800))), name = tensor("decoder_layers_5_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_5_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_5_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(348805888)))]; + tensor decoder_layers_5_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(348810048))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(349858688))), name = tensor("decoder_layers_5_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_5_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_5_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(349859776)))]; + tensor decoder_layers_5_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(349863936))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(350912576))), name = tensor("decoder_layers_5_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_5_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_5_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(350913664)))]; + tensor decoder_layers_5_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(350917824))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(351966464))), name = tensor("decoder_layers_5_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_5_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_5_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(351967552)))]; + tensor decoder_layers_5_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_5_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(351971712)))]; + tensor decoder_layers_5_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_5_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(351975872)))]; + tensor decoder_layers_5_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(351980032))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(353028672))), name = tensor("decoder_layers_5_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_5_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_5_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(353029760)))]; + tensor decoder_layers_5_encoder_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(353033920))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(354082560))), name = tensor("decoder_layers_5_encoder_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_5_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_5_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(354083648)))]; + tensor decoder_layers_5_encoder_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(354087808))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(355136448))), name = tensor("decoder_layers_5_encoder_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_5_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_5_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(355137536)))]; + tensor decoder_layers_5_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(355141696))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(356190336))), name = tensor("decoder_layers_5_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_5_final_layer_norm_bias = const()[name = tensor("decoder_layers_5_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(356191424)))]; + tensor decoder_layers_5_final_layer_norm_weight = const()[name = tensor("decoder_layers_5_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(356195584)))]; + tensor decoder_layers_5_fc1_bias = const()[name = tensor("decoder_layers_5_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(356199744)))]; + tensor decoder_layers_5_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(356216192))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(360410560))), name = tensor("decoder_layers_5_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_5_fc2_bias = const()[name = tensor("decoder_layers_5_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(360411648)))]; + tensor decoder_layers_5_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(360415808))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(364610176))), name = tensor("decoder_layers_5_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_6_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_6_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(364611264)))]; + tensor decoder_layers_6_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_6_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(364615424)))]; + tensor decoder_layers_6_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_6_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(364619584)))]; + tensor decoder_layers_6_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(364623744))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(365672384))), name = tensor("decoder_layers_6_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_6_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_6_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(365673472)))]; + tensor decoder_layers_6_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(365677632))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(366726272))), name = tensor("decoder_layers_6_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_6_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_6_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(366727360)))]; + tensor decoder_layers_6_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(366731520))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(367780160))), name = tensor("decoder_layers_6_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_6_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_6_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(367781248)))]; + tensor decoder_layers_6_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(367785408))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(368834048))), name = tensor("decoder_layers_6_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_6_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_6_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(368835136)))]; + tensor decoder_layers_6_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_6_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(368839296)))]; + tensor decoder_layers_6_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_6_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(368843456)))]; + tensor decoder_layers_6_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(368847616))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(369896256))), name = tensor("decoder_layers_6_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_6_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_6_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(369897344)))]; + tensor decoder_layers_6_encoder_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(369901504))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(370950144))), name = tensor("decoder_layers_6_encoder_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_6_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_6_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(370951232)))]; + tensor decoder_layers_6_encoder_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(370955392))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(372004032))), name = tensor("decoder_layers_6_encoder_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_6_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_6_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(372005120)))]; + tensor decoder_layers_6_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(372009280))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(373057920))), name = tensor("decoder_layers_6_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_6_final_layer_norm_bias = const()[name = tensor("decoder_layers_6_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(373059008)))]; + tensor decoder_layers_6_final_layer_norm_weight = const()[name = tensor("decoder_layers_6_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(373063168)))]; + tensor decoder_layers_6_fc1_bias = const()[name = tensor("decoder_layers_6_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(373067328)))]; + tensor decoder_layers_6_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(373083776))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(377278144))), name = tensor("decoder_layers_6_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_6_fc2_bias = const()[name = tensor("decoder_layers_6_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(377279232)))]; + tensor decoder_layers_6_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(377283392))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(381477760))), name = tensor("decoder_layers_6_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_7_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_7_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(381478848)))]; + tensor decoder_layers_7_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_7_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(381483008)))]; + tensor decoder_layers_7_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_7_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(381487168)))]; + tensor decoder_layers_7_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(381491328))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(382539968))), name = tensor("decoder_layers_7_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_7_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_7_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(382541056)))]; + tensor decoder_layers_7_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(382545216))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(383593856))), name = tensor("decoder_layers_7_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_7_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_7_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(383594944)))]; + tensor decoder_layers_7_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(383599104))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(384647744))), name = tensor("decoder_layers_7_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_7_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_7_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(384648832)))]; + tensor decoder_layers_7_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(384652992))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(385701632))), name = tensor("decoder_layers_7_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_7_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_7_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(385702720)))]; + tensor decoder_layers_7_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_7_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(385706880)))]; + tensor decoder_layers_7_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_7_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(385711040)))]; + tensor decoder_layers_7_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(385715200))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(386763840))), name = tensor("decoder_layers_7_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_7_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_7_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(386764928)))]; + tensor decoder_layers_7_encoder_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(386769088))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(387817728))), name = tensor("decoder_layers_7_encoder_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_7_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_7_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(387818816)))]; + tensor decoder_layers_7_encoder_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(387822976))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(388871616))), name = tensor("decoder_layers_7_encoder_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_7_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_7_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(388872704)))]; + tensor decoder_layers_7_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(388876864))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(377278144))), name = tensor("decoder_layers_7_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_7_final_layer_norm_bias = const()[name = tensor("decoder_layers_7_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(389925504)))]; + tensor decoder_layers_7_final_layer_norm_weight = const()[name = tensor("decoder_layers_7_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(389929664)))]; + tensor decoder_layers_7_fc1_bias = const()[name = tensor("decoder_layers_7_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(389933824)))]; + tensor decoder_layers_7_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(389950272))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263406848))), name = tensor("decoder_layers_7_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_7_fc2_bias = const()[name = tensor("decoder_layers_7_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(394144640)))]; + tensor decoder_layers_7_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(394148800))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(398343168))), name = tensor("decoder_layers_7_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_8_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_8_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(398344256)))]; + tensor decoder_layers_8_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_8_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(398348416)))]; + tensor decoder_layers_8_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_8_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(398352576)))]; + tensor decoder_layers_8_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(398356736))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(399405376))), name = tensor("decoder_layers_8_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_8_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_8_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(399406464)))]; + tensor decoder_layers_8_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(399410624))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(400459264))), name = tensor("decoder_layers_8_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_8_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_8_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(400460352)))]; + tensor decoder_layers_8_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(400464512))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263406848))), name = tensor("decoder_layers_8_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_8_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_8_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(401513152)))]; + tensor decoder_layers_8_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(401517312))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(402565952))), name = tensor("decoder_layers_8_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_8_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_8_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(402567040)))]; + tensor decoder_layers_8_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_8_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(402571200)))]; + tensor decoder_layers_8_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_8_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(402575360)))]; + tensor decoder_layers_8_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(402579520))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(403628160))), name = tensor("decoder_layers_8_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_8_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_8_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(403629248)))]; + tensor decoder_layers_8_encoder_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(403633408))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(404682048))), name = tensor("decoder_layers_8_encoder_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_8_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_8_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(404683136)))]; + tensor decoder_layers_8_encoder_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(404687296))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(405735936))), name = tensor("decoder_layers_8_encoder_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_8_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_8_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(405737024)))]; + tensor decoder_layers_8_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(405741184))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(406789824))), name = tensor("decoder_layers_8_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_8_final_layer_norm_bias = const()[name = tensor("decoder_layers_8_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(406790912)))]; + tensor decoder_layers_8_final_layer_norm_weight = const()[name = tensor("decoder_layers_8_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(406795072)))]; + tensor decoder_layers_8_fc1_bias = const()[name = tensor("decoder_layers_8_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(406799232)))]; + tensor decoder_layers_8_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(406815680))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(411010048))), name = tensor("decoder_layers_8_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_8_fc2_bias = const()[name = tensor("decoder_layers_8_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(411011136)))]; + tensor decoder_layers_8_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(411015296))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(415209664))), name = tensor("decoder_layers_8_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_9_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_9_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(415210752)))]; + tensor decoder_layers_9_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_9_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(415214912)))]; + tensor decoder_layers_9_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_9_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(415219072)))]; + tensor decoder_layers_9_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(415223232))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(416271872))), name = tensor("decoder_layers_9_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_9_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_9_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(416272960)))]; + tensor decoder_layers_9_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(416277120))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263406848))), name = tensor("decoder_layers_9_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_9_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_9_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(417325760)))]; + tensor decoder_layers_9_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(417329920))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(418378560))), name = tensor("decoder_layers_9_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_9_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_9_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(418379648)))]; + tensor decoder_layers_9_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(418383808))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(419432448))), name = tensor("decoder_layers_9_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_9_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_9_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(419433536)))]; + tensor decoder_layers_9_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_9_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(419437696)))]; + tensor decoder_layers_9_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_9_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(419441856)))]; + tensor decoder_layers_9_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(419446016))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(420494656))), name = tensor("decoder_layers_9_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_9_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_9_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(420495744)))]; + tensor decoder_layers_9_encoder_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(420499904))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(421548544))), name = tensor("decoder_layers_9_encoder_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_9_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_9_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(421549632)))]; + tensor decoder_layers_9_encoder_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(421553792))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263406848))), name = tensor("decoder_layers_9_encoder_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_9_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_9_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(422602432)))]; + tensor decoder_layers_9_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(422606592))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(423655232))), name = tensor("decoder_layers_9_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_9_final_layer_norm_bias = const()[name = tensor("decoder_layers_9_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(423656320)))]; + tensor decoder_layers_9_final_layer_norm_weight = const()[name = tensor("decoder_layers_9_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(423660480)))]; + tensor decoder_layers_9_fc1_bias = const()[name = tensor("decoder_layers_9_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(423664640)))]; + tensor decoder_layers_9_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(423681088))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(427875456))), name = tensor("decoder_layers_9_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_9_fc2_bias = const()[name = tensor("decoder_layers_9_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(427876544)))]; + tensor decoder_layers_9_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(427880704))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(432075072))), name = tensor("decoder_layers_9_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_10_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_10_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(432076160)))]; + tensor decoder_layers_10_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_10_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(432080320)))]; + tensor decoder_layers_10_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_10_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(432084480)))]; + tensor decoder_layers_10_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(432088640))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(433137280))), name = tensor("decoder_layers_10_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_10_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_10_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(433138368)))]; + tensor decoder_layers_10_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(433142528))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(434191168))), name = tensor("decoder_layers_10_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_10_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_10_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(434192256)))]; + tensor decoder_layers_10_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(434196416))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(435245056))), name = tensor("decoder_layers_10_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_10_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_10_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(435246144)))]; + tensor decoder_layers_10_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(435250304))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(436298944))), name = tensor("decoder_layers_10_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_10_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_10_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(436300032)))]; + tensor decoder_layers_10_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_10_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(436304192)))]; + tensor decoder_layers_10_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_10_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(436308352)))]; + tensor decoder_layers_10_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(436312512))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(437361152))), name = tensor("decoder_layers_10_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_10_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_10_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(437362240)))]; + tensor decoder_layers_10_encoder_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(437366400))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(438415040))), name = tensor("decoder_layers_10_encoder_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_10_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_10_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(438416128)))]; + tensor decoder_layers_10_encoder_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(438420288))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263406848))), name = tensor("decoder_layers_10_encoder_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_10_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_10_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(439468928)))]; + tensor decoder_layers_10_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(439473088))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(440521728))), name = tensor("decoder_layers_10_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_10_final_layer_norm_bias = const()[name = tensor("decoder_layers_10_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(440522816)))]; + tensor decoder_layers_10_final_layer_norm_weight = const()[name = tensor("decoder_layers_10_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(440526976)))]; + tensor decoder_layers_10_fc1_bias = const()[name = tensor("decoder_layers_10_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(440531136)))]; + tensor decoder_layers_10_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(440547584))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(444741952))), name = tensor("decoder_layers_10_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_10_fc2_bias = const()[name = tensor("decoder_layers_10_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(444743040)))]; + tensor decoder_layers_10_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(444747200))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(448941568))), name = tensor("decoder_layers_10_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layers_11_self_attn_layer_norm_bias = const()[name = tensor("decoder_layers_11_self_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(448942656)))]; + tensor decoder_layers_11_self_attn_layer_norm_weight = const()[name = tensor("decoder_layers_11_self_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(448946816)))]; + tensor decoder_layers_11_self_attn_q_proj_bias = const()[name = tensor("decoder_layers_11_self_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(448950976)))]; + tensor decoder_layers_11_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(448955136))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(450003776))), name = tensor("decoder_layers_11_self_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_11_self_attn_k_proj_bias = const()[name = tensor("decoder_layers_11_self_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(450004864)))]; + tensor decoder_layers_11_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(450009024))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(451057664))), name = tensor("decoder_layers_11_self_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_11_self_attn_v_proj_bias = const()[name = tensor("decoder_layers_11_self_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(451058752)))]; + tensor decoder_layers_11_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(451062912))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(452111552))), name = tensor("decoder_layers_11_self_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_11_self_attn_out_proj_bias = const()[name = tensor("decoder_layers_11_self_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(452112640)))]; + tensor decoder_layers_11_self_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(452116800))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(453165440))), name = tensor("decoder_layers_11_self_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_11_encoder_attn_layer_norm_bias = const()[name = tensor("decoder_layers_11_encoder_attn_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(453166528)))]; + tensor decoder_layers_11_encoder_attn_layer_norm_weight = const()[name = tensor("decoder_layers_11_encoder_attn_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(453170688)))]; + tensor decoder_layers_11_encoder_attn_q_proj_bias = const()[name = tensor("decoder_layers_11_encoder_attn_q_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(453174848)))]; + tensor decoder_layers_11_encoder_attn_q_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(453179008))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(454227648))), name = tensor("decoder_layers_11_encoder_attn_q_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_11_encoder_attn_k_proj_bias = const()[name = tensor("decoder_layers_11_encoder_attn_k_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(454228736)))]; + tensor decoder_layers_11_encoder_attn_k_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(454232896))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(455281536))), name = tensor("decoder_layers_11_encoder_attn_k_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_11_encoder_attn_v_proj_bias = const()[name = tensor("decoder_layers_11_encoder_attn_v_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(455282624)))]; + tensor decoder_layers_11_encoder_attn_v_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(455286784))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(263406848))), name = tensor("decoder_layers_11_encoder_attn_v_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_11_encoder_attn_out_proj_bias = const()[name = tensor("decoder_layers_11_encoder_attn_out_proj_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(456335424)))]; + tensor decoder_layers_11_encoder_attn_out_proj_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(456339584))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(457388224))), name = tensor("decoder_layers_11_encoder_attn_out_proj_weight_palettized"), shape = tensor([1024, 1024])]; + tensor decoder_layers_11_final_layer_norm_bias = const()[name = tensor("decoder_layers_11_final_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(457389312)))]; + tensor decoder_layers_11_final_layer_norm_weight = const()[name = tensor("decoder_layers_11_final_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(457393472)))]; + tensor decoder_layers_11_fc1_bias = const()[name = tensor("decoder_layers_11_fc1_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(457397632)))]; + tensor decoder_layers_11_fc1_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(457414080))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(461608448))), name = tensor("decoder_layers_11_fc1_weight_palettized"), shape = tensor([4096, 1024])]; + tensor decoder_layers_11_fc2_bias = const()[name = tensor("decoder_layers_11_fc2_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(461609536)))]; + tensor decoder_layers_11_fc2_weight_palettized = constexpr_lut_to_dense()[indices = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(461613696))), lut = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(284498304))), name = tensor("decoder_layers_11_fc2_weight_palettized"), shape = tensor([1024, 4096])]; + tensor decoder_layer_norm_bias = const()[name = tensor("decoder_layer_norm_bias"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(465808064)))]; + tensor decoder_layer_norm_weight = const()[name = tensor("decoder_layer_norm_weight"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(465812224)))]; + tensor var_9 = const()[name = tensor("op_9"), val = tensor(0x1.4f8b58p-17)]; + tensor var_11 = const()[name = tensor("op_11"), val = tensor(0x1p-3)]; + tensor var_13 = const()[name = tensor("op_13"), val = tensor(-2)]; + tensor var_24 = const()[name = tensor("op_24"), val = tensor(-0x1.fffffep+127)]; + tensor var_27 = const()[name = tensor("op_27"), val = tensor(0)]; + tensor var_30 = const()[name = tensor("op_30"), val = tensor(1)]; + tensor const_0 = const()[name = tensor("const_0"), val = tensor(2)]; + tensor var_62_axis_0 = const()[name = tensor("op_62_axis_0"), val = tensor(0)]; + tensor var_62_batch_dims_0 = const()[name = tensor("op_62_batch_dims_0"), val = tensor(0)]; + tensor var_62 = gather(axis = var_62_axis_0, batch_dims = var_62_batch_dims_0, indices = input_ids, x = decoder_embed_tokens_weight_palettized)[name = tensor("op_62")]; + tensor var_63 = const()[name = tensor("op_63"), val = tensor(0x1p+5)]; + tensor inputs_embeds = mul(x = var_62, y = var_63)[name = tensor("inputs_embeds")]; + tensor shape_1 = const()[name = tensor("shape_1"), val = tensor([1, 1, 2, 2])]; + tensor reshape_1 = const()[name = tensor("reshape_1"), val = tensor([0, 1, 2, 3])]; + tensor reshape_2 = const()[name = tensor("reshape_2"), val = tensor([0x0p+0, -0x1.fffffep+127, 0x0p+0, 0x0p+0])]; + tensor reshape_3 = const()[name = tensor("reshape_3"), val = tensor([0x0p+0, -0x1.fffffep+127, 0x0p+0, 0x0p+0])]; + tensor scatter_0_mode_0 = const()[name = tensor("scatter_0_mode_0"), val = tensor("update")]; + tensor scatter_0_axis_0 = const()[name = tensor("scatter_0_axis_0"), val = tensor(0)]; + tensor scatter_0 = scatter(axis = scatter_0_axis_0, data = reshape_3, indices = reshape_1, mode = scatter_0_mode_0, updates = reshape_2)[name = tensor("scatter_0")]; + tensor reshape_4 = reshape(shape = shape_1, x = scatter_0)[name = tensor("reshape_4")]; + tensor var_117_shape = shape(x = encoder_attention_mask)[name = tensor("op_117_shape")]; + tensor gather_0 = const()[name = tensor("gather_0"), val = tensor(1)]; + tensor gather_1_indices_0 = const()[name = tensor("gather_1_indices_0"), val = tensor(1)]; + tensor gather_1_axis_0 = const()[name = tensor("gather_1_axis_0"), val = tensor(0)]; + tensor gather_1_batch_dims_0 = const()[name = tensor("gather_1_batch_dims_0"), val = tensor(0)]; + tensor gather_1 = gather(axis = gather_1_axis_0, batch_dims = gather_1_batch_dims_0, indices = gather_1_indices_0, x = var_117_shape)[name = tensor("gather_1")]; + tensor var_120_axes_0 = const()[name = tensor("op_120_axes_0"), val = tensor([1])]; + tensor var_120 = expand_dims(axes = var_120_axes_0, x = encoder_attention_mask)[name = tensor("op_120")]; + tensor var_121_axes_0 = const()[name = tensor("op_121_axes_0"), val = tensor([2])]; + tensor var_121 = expand_dims(axes = var_121_axes_0, x = var_120)[name = tensor("op_121")]; + tensor concat_3_axis_0 = const()[name = tensor("concat_3_axis_0"), val = tensor(0)]; + tensor concat_3_interleave_0 = const()[name = tensor("concat_3_interleave_0"), val = tensor(false)]; + tensor concat_3 = concat(axis = concat_3_axis_0, interleave = concat_3_interleave_0, values = (gather_0, var_30, const_0, gather_1))[name = tensor("concat_3")]; + tensor shape_0 = shape(x = var_121)[name = tensor("shape_0")]; + tensor equal_0_y_0 = const()[name = tensor("equal_0_y_0"), val = tensor(-1)]; + tensor equal_0 = equal(x = concat_3, y = equal_0_y_0)[name = tensor("equal_0")]; + tensor select_0 = select(a = shape_0, b = concat_3, cond = equal_0)[name = tensor("select_0")]; + tensor real_div_0 = real_div(x = select_0, y = shape_0)[name = tensor("real_div_0")]; + tensor var_124 = tile(reps = real_div_0, x = var_121)[name = tensor("op_124")]; + tensor expanded_mask_dtype_0 = const()[name = tensor("expanded_mask_dtype_0"), val = tensor("fp32")]; + tensor const_11 = const()[name = tensor("const_11"), val = tensor(0x1p+0)]; + tensor expanded_mask = cast(dtype = expanded_mask_dtype_0, x = var_124)[name = tensor("cast_2")]; + tensor inverted_mask = sub(x = const_11, y = expanded_mask)[name = tensor("inverted_mask")]; + tensor var_129_dtype_0 = const()[name = tensor("op_129_dtype_0"), val = tensor("bool")]; + tensor var_129 = cast(dtype = var_129_dtype_0, x = inverted_mask)[name = tensor("cast_1")]; + tensor attention_mask_5 = select(a = var_24, b = inverted_mask, cond = var_129)[name = tensor("attention_mask_5")]; + tensor var_134 = not_equal(x = input_ids, y = var_30)[name = tensor("op_134")]; + tensor mask_dtype_0 = const()[name = tensor("mask_dtype_0"), val = tensor("int32")]; + tensor var_136_exclusive_0 = const()[name = tensor("op_136_exclusive_0"), val = tensor(false)]; + tensor var_136_reverse_0 = const()[name = tensor("op_136_reverse_0"), val = tensor(false)]; + tensor mask = cast(dtype = mask_dtype_0, x = var_134)[name = tensor("cast_0")]; + tensor var_136 = cumsum(axis = var_30, exclusive = var_136_exclusive_0, reverse = var_136_reverse_0, x = mask)[name = tensor("op_136")]; + tensor incremental_indices = mul(x = var_136, y = mask)[name = tensor("incremental_indices")]; + tensor var_142 = const()[name = tensor("op_142"), val = tensor(1)]; + tensor var_143 = add(x = incremental_indices, y = var_142)[name = tensor("op_143")]; + tensor var_145 = const()[name = tensor("op_145"), val = tensor([-1])]; + tensor var_146 = reshape(shape = var_145, x = var_143)[name = tensor("op_146")]; + tensor var_147_batch_dims_0 = const()[name = tensor("op_147_batch_dims_0"), val = tensor(0)]; + tensor var_147 = gather(axis = var_27, batch_dims = var_147_batch_dims_0, indices = var_146, x = decoder_embed_positions_weights_palettized)[name = tensor("op_147")]; + tensor var_149 = const()[name = tensor("op_149"), val = tensor([1, 2, 1024])]; + tensor var_150 = reshape(shape = var_149, x = var_147)[name = tensor("op_150")]; + tensor input_3 = add(x = inputs_embeds, y = var_150)[name = tensor("input_3")]; + tensor hidden_states_1_axes_0 = const()[name = tensor("hidden_states_1_axes_0"), val = tensor([-1])]; + tensor hidden_states_1 = layer_norm(axes = hidden_states_1_axes_0, beta = decoder_layers_0_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_0_self_attn_layer_norm_weight, x = input_3)[name = tensor("hidden_states_1")]; + tensor var_174 = linear(bias = decoder_layers_0_self_attn_q_proj_bias, weight = decoder_layers_0_self_attn_q_proj_weight_palettized, x = hidden_states_1)[name = tensor("linear_0")]; + tensor var_175 = const()[name = tensor("op_175"), val = tensor([1, 2, -1, 64])]; + tensor var_176 = reshape(shape = var_175, x = var_174)[name = tensor("op_176")]; + tensor key_states_1 = linear(bias = decoder_layers_0_self_attn_k_proj_bias, weight = decoder_layers_0_self_attn_k_proj_weight_palettized, x = hidden_states_1)[name = tensor("linear_1")]; + tensor value_states_1 = linear(bias = decoder_layers_0_self_attn_v_proj_bias, weight = decoder_layers_0_self_attn_v_proj_weight_palettized, x = hidden_states_1)[name = tensor("linear_2")]; + tensor var_184 = const()[name = tensor("op_184"), val = tensor([1, 2, -1, 64])]; + tensor var_185 = reshape(shape = var_184, x = key_states_1)[name = tensor("op_185")]; + tensor var_187 = const()[name = tensor("op_187"), val = tensor([1, 2, -1, 64])]; + tensor var_188 = reshape(shape = var_187, x = value_states_1)[name = tensor("op_188")]; + tensor value_states_3_perm_0 = const()[name = tensor("value_states_3_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_1_interleave_0 = const()[name = tensor("key_1_interleave_0"), val = tensor(false)]; + tensor const_123 = const()[name = tensor("const_123"), val = tensor(1)]; + tensor key_1 = concat(axis = const_123, interleave = key_1_interleave_0, values = var_185)[name = tensor("key_1")]; + tensor value_1_interleave_0 = const()[name = tensor("value_1_interleave_0"), val = tensor(false)]; + tensor value_states_3 = transpose(perm = value_states_3_perm_0, x = var_188)[name = tensor("transpose_215")]; + tensor value_1 = concat(axis = var_13, interleave = value_1_interleave_0, values = value_states_3)[name = tensor("value_1")]; + tensor mul_0 = mul(x = var_176, y = var_11)[name = tensor("mul_0")]; + tensor matmul_0_transpose_y_0 = const()[name = tensor("matmul_0_transpose_y_0"), val = tensor(true)]; + tensor matmul_0_transpose_x_0 = const()[name = tensor("matmul_0_transpose_x_0"), val = tensor(false)]; + tensor transpose_72_perm_0 = const()[name = tensor("transpose_72_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_73_perm_0 = const()[name = tensor("transpose_73_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_73 = transpose(perm = transpose_73_perm_0, x = key_1)[name = tensor("transpose_213")]; + tensor transpose_72 = transpose(perm = transpose_72_perm_0, x = mul_0)[name = tensor("transpose_214")]; + tensor matmul_0 = matmul(transpose_x = matmul_0_transpose_x_0, transpose_y = matmul_0_transpose_y_0, x = transpose_72, y = transpose_73)[name = tensor("matmul_0")]; + tensor add_0 = add(x = matmul_0, y = reshape_4)[name = tensor("add_0")]; + tensor softmax_0_axis_0 = const()[name = tensor("softmax_0_axis_0"), val = tensor(-1)]; + tensor softmax_0 = softmax(axis = softmax_0_axis_0, x = add_0)[name = tensor("softmax_0")]; + tensor attn_output_1_transpose_x_0 = const()[name = tensor("attn_output_1_transpose_x_0"), val = tensor(false)]; + tensor attn_output_1_transpose_y_0 = const()[name = tensor("attn_output_1_transpose_y_0"), val = tensor(false)]; + tensor attn_output_1 = matmul(transpose_x = attn_output_1_transpose_x_0, transpose_y = attn_output_1_transpose_y_0, x = softmax_0, y = value_1)[name = tensor("attn_output_1")]; + tensor var_204_perm_0 = const()[name = tensor("op_204_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_206 = const()[name = tensor("op_206"), val = tensor([1, 2, -1])]; + tensor var_204 = transpose(perm = var_204_perm_0, x = attn_output_1)[name = tensor("transpose_212")]; + tensor var_207 = reshape(shape = var_206, x = var_204)[name = tensor("op_207")]; + tensor input_9 = linear(bias = decoder_layers_0_self_attn_out_proj_bias, weight = decoder_layers_0_self_attn_out_proj_weight_palettized, x = var_207)[name = tensor("linear_3")]; + tensor input_11 = add(x = input_3, y = input_9)[name = tensor("input_11")]; + tensor hidden_states_5_axes_0 = const()[name = tensor("hidden_states_5_axes_0"), val = tensor([-1])]; + tensor hidden_states_5 = layer_norm(axes = hidden_states_5_axes_0, beta = decoder_layers_0_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_0_encoder_attn_layer_norm_weight, x = input_11)[name = tensor("hidden_states_5")]; + tensor var_231 = linear(bias = decoder_layers_0_encoder_attn_q_proj_bias, weight = decoder_layers_0_encoder_attn_q_proj_weight_palettized, x = hidden_states_5)[name = tensor("linear_4")]; + tensor var_232 = const()[name = tensor("op_232"), val = tensor([1, 2, -1, 64])]; + tensor var_233 = reshape(shape = var_232, x = var_231)[name = tensor("op_233")]; + tensor query_3_perm_0 = const()[name = tensor("query_3_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_5 = linear(bias = decoder_layers_0_encoder_attn_k_proj_bias, weight = decoder_layers_0_encoder_attn_k_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_5")]; + tensor value_states_5 = linear(bias = decoder_layers_0_encoder_attn_v_proj_bias, weight = decoder_layers_0_encoder_attn_v_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_6")]; + tensor concat_4x = const()[name = tensor("concat_4x"), val = tensor([1, -1, 16, 64])]; + tensor var_242 = reshape(shape = concat_4x, x = key_states_5)[name = tensor("op_242")]; + tensor key_states_7_perm_0 = const()[name = tensor("key_states_7_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_5x = const()[name = tensor("concat_5x"), val = tensor([1, -1, 16, 64])]; + tensor var_245 = reshape(shape = concat_5x, x = value_states_5)[name = tensor("op_245")]; + tensor value_states_7_perm_0 = const()[name = tensor("value_states_7_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_3_interleave_0 = const()[name = tensor("key_3_interleave_0"), val = tensor(false)]; + tensor key_states_7 = transpose(perm = key_states_7_perm_0, x = var_242)[name = tensor("transpose_211")]; + tensor key_3 = concat(axis = var_13, interleave = key_3_interleave_0, values = key_states_7)[name = tensor("key_3")]; + tensor value_3_interleave_0 = const()[name = tensor("value_3_interleave_0"), val = tensor(false)]; + tensor value_states_7 = transpose(perm = value_states_7_perm_0, x = var_245)[name = tensor("transpose_210")]; + tensor value_3 = concat(axis = var_13, interleave = value_3_interleave_0, values = value_states_7)[name = tensor("value_3")]; + tensor var_255_shape = shape(x = key_3)[name = tensor("op_255_shape")]; + tensor gather_3_indices_0 = const()[name = tensor("gather_3_indices_0"), val = tensor(2)]; + tensor gather_3_axis_0 = const()[name = tensor("gather_3_axis_0"), val = tensor(0)]; + tensor gather_3_batch_dims_0 = const()[name = tensor("gather_3_batch_dims_0"), val = tensor(0)]; + tensor gather_3 = gather(axis = gather_3_axis_0, batch_dims = gather_3_batch_dims_0, indices = gather_3_indices_0, x = var_255_shape)[name = tensor("gather_3")]; + tensor concat_6_values0_0 = const()[name = tensor("concat_6_values0_0"), val = tensor(0)]; + tensor concat_6_values1_0 = const()[name = tensor("concat_6_values1_0"), val = tensor(0)]; + tensor concat_6_values2_0 = const()[name = tensor("concat_6_values2_0"), val = tensor(0)]; + tensor concat_6_axis_0 = const()[name = tensor("concat_6_axis_0"), val = tensor(0)]; + tensor concat_6_interleave_0 = const()[name = tensor("concat_6_interleave_0"), val = tensor(false)]; + tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (concat_6_values0_0, concat_6_values1_0, concat_6_values2_0, gather_3))[name = tensor("concat_6")]; + tensor attention_mask_7_begin_0 = const()[name = tensor("attention_mask_7_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_7_end_mask_0 = const()[name = tensor("attention_mask_7_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_7 = slice_by_index(begin = attention_mask_7_begin_0, end = concat_6, end_mask = attention_mask_7_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_7")]; + tensor query_3 = transpose(perm = query_3_perm_0, x = var_233)[name = tensor("transpose_209")]; + tensor mul_1 = mul(x = query_3, y = var_11)[name = tensor("mul_1")]; + tensor matmul_1_transpose_y_0 = const()[name = tensor("matmul_1_transpose_y_0"), val = tensor(true)]; + tensor matmul_1_transpose_x_0 = const()[name = tensor("matmul_1_transpose_x_0"), val = tensor(false)]; + tensor matmul_1 = matmul(transpose_x = matmul_1_transpose_x_0, transpose_y = matmul_1_transpose_y_0, x = mul_1, y = key_3)[name = tensor("matmul_1")]; + tensor add_1 = add(x = matmul_1, y = attention_mask_7)[name = tensor("add_1")]; + tensor softmax_1_axis_0 = const()[name = tensor("softmax_1_axis_0"), val = tensor(-1)]; + tensor softmax_1 = softmax(axis = softmax_1_axis_0, x = add_1)[name = tensor("softmax_1")]; + tensor attn_output_5_transpose_x_0 = const()[name = tensor("attn_output_5_transpose_x_0"), val = tensor(false)]; + tensor attn_output_5_transpose_y_0 = const()[name = tensor("attn_output_5_transpose_y_0"), val = tensor(false)]; + tensor attn_output_5 = matmul(transpose_x = attn_output_5_transpose_x_0, transpose_y = attn_output_5_transpose_y_0, x = softmax_1, y = value_3)[name = tensor("attn_output_5")]; + tensor var_261_perm_0 = const()[name = tensor("op_261_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_263 = const()[name = tensor("op_263"), val = tensor([1, 2, -1])]; + tensor var_261 = transpose(perm = var_261_perm_0, x = attn_output_5)[name = tensor("transpose_208")]; + tensor var_264 = reshape(shape = var_263, x = var_261)[name = tensor("op_264")]; + tensor input_15 = linear(bias = decoder_layers_0_encoder_attn_out_proj_bias, weight = decoder_layers_0_encoder_attn_out_proj_weight_palettized, x = var_264)[name = tensor("linear_7")]; + tensor input_17 = add(x = input_11, y = input_15)[name = tensor("input_17")]; + tensor input_19_axes_0 = const()[name = tensor("input_19_axes_0"), val = tensor([-1])]; + tensor input_19 = layer_norm(axes = input_19_axes_0, beta = decoder_layers_0_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_0_final_layer_norm_weight, x = input_17)[name = tensor("input_19")]; + tensor input_21 = linear(bias = decoder_layers_0_fc1_bias, weight = decoder_layers_0_fc1_weight_palettized, x = input_19)[name = tensor("linear_8")]; + tensor input_23 = relu(x = input_21)[name = tensor("input_23")]; + tensor input_27 = linear(bias = decoder_layers_0_fc2_bias, weight = decoder_layers_0_fc2_weight_palettized, x = input_23)[name = tensor("linear_9")]; + tensor input_29 = add(x = input_17, y = input_27)[name = tensor("input_29")]; + tensor hidden_states_11_axes_0 = const()[name = tensor("hidden_states_11_axes_0"), val = tensor([-1])]; + tensor hidden_states_11 = layer_norm(axes = hidden_states_11_axes_0, beta = decoder_layers_1_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_1_self_attn_layer_norm_weight, x = input_29)[name = tensor("hidden_states_11")]; + tensor var_314 = linear(bias = decoder_layers_1_self_attn_q_proj_bias, weight = decoder_layers_1_self_attn_q_proj_weight_palettized, x = hidden_states_11)[name = tensor("linear_10")]; + tensor var_315 = const()[name = tensor("op_315"), val = tensor([1, 2, -1, 64])]; + tensor var_316 = reshape(shape = var_315, x = var_314)[name = tensor("op_316")]; + tensor key_states_9 = linear(bias = decoder_layers_1_self_attn_k_proj_bias, weight = decoder_layers_1_self_attn_k_proj_weight_palettized, x = hidden_states_11)[name = tensor("linear_11")]; + tensor value_states_9 = linear(bias = decoder_layers_1_self_attn_v_proj_bias, weight = decoder_layers_1_self_attn_v_proj_weight_palettized, x = hidden_states_11)[name = tensor("linear_12")]; + tensor var_324 = const()[name = tensor("op_324"), val = tensor([1, 2, -1, 64])]; + tensor var_325 = reshape(shape = var_324, x = key_states_9)[name = tensor("op_325")]; + tensor var_327 = const()[name = tensor("op_327"), val = tensor([1, 2, -1, 64])]; + tensor var_328 = reshape(shape = var_327, x = value_states_9)[name = tensor("op_328")]; + tensor value_states_11_perm_0 = const()[name = tensor("value_states_11_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_5_interleave_0 = const()[name = tensor("key_5_interleave_0"), val = tensor(false)]; + tensor const_124 = const()[name = tensor("const_124"), val = tensor(1)]; + tensor key_5 = concat(axis = const_124, interleave = key_5_interleave_0, values = var_325)[name = tensor("key_5")]; + tensor value_5_interleave_0 = const()[name = tensor("value_5_interleave_0"), val = tensor(false)]; + tensor value_states_11 = transpose(perm = value_states_11_perm_0, x = var_328)[name = tensor("transpose_207")]; + tensor value_5 = concat(axis = var_13, interleave = value_5_interleave_0, values = value_states_11)[name = tensor("value_5")]; + tensor mul_2 = mul(x = var_316, y = var_11)[name = tensor("mul_2")]; + tensor matmul_2_transpose_y_0 = const()[name = tensor("matmul_2_transpose_y_0"), val = tensor(true)]; + tensor matmul_2_transpose_x_0 = const()[name = tensor("matmul_2_transpose_x_0"), val = tensor(false)]; + tensor transpose_74_perm_0 = const()[name = tensor("transpose_74_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_75_perm_0 = const()[name = tensor("transpose_75_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_75 = transpose(perm = transpose_75_perm_0, x = key_5)[name = tensor("transpose_205")]; + tensor transpose_74 = transpose(perm = transpose_74_perm_0, x = mul_2)[name = tensor("transpose_206")]; + tensor matmul_2 = matmul(transpose_x = matmul_2_transpose_x_0, transpose_y = matmul_2_transpose_y_0, x = transpose_74, y = transpose_75)[name = tensor("matmul_2")]; + tensor add_2 = add(x = matmul_2, y = reshape_4)[name = tensor("add_2")]; + tensor softmax_2_axis_0 = const()[name = tensor("softmax_2_axis_0"), val = tensor(-1)]; + tensor softmax_2 = softmax(axis = softmax_2_axis_0, x = add_2)[name = tensor("softmax_2")]; + tensor attn_output_9_transpose_x_0 = const()[name = tensor("attn_output_9_transpose_x_0"), val = tensor(false)]; + tensor attn_output_9_transpose_y_0 = const()[name = tensor("attn_output_9_transpose_y_0"), val = tensor(false)]; + tensor attn_output_9 = matmul(transpose_x = attn_output_9_transpose_x_0, transpose_y = attn_output_9_transpose_y_0, x = softmax_2, y = value_5)[name = tensor("attn_output_9")]; + tensor var_344_perm_0 = const()[name = tensor("op_344_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_346 = const()[name = tensor("op_346"), val = tensor([1, 2, -1])]; + tensor var_344 = transpose(perm = var_344_perm_0, x = attn_output_9)[name = tensor("transpose_204")]; + tensor var_347 = reshape(shape = var_346, x = var_344)[name = tensor("op_347")]; + tensor input_33 = linear(bias = decoder_layers_1_self_attn_out_proj_bias, weight = decoder_layers_1_self_attn_out_proj_weight_palettized, x = var_347)[name = tensor("linear_13")]; + tensor input_35 = add(x = input_29, y = input_33)[name = tensor("input_35")]; + tensor hidden_states_15_axes_0 = const()[name = tensor("hidden_states_15_axes_0"), val = tensor([-1])]; + tensor hidden_states_15 = layer_norm(axes = hidden_states_15_axes_0, beta = decoder_layers_1_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_1_encoder_attn_layer_norm_weight, x = input_35)[name = tensor("hidden_states_15")]; + tensor var_371 = linear(bias = decoder_layers_1_encoder_attn_q_proj_bias, weight = decoder_layers_1_encoder_attn_q_proj_weight_palettized, x = hidden_states_15)[name = tensor("linear_14")]; + tensor var_372 = const()[name = tensor("op_372"), val = tensor([1, 2, -1, 64])]; + tensor var_373 = reshape(shape = var_372, x = var_371)[name = tensor("op_373")]; + tensor query_7_perm_0 = const()[name = tensor("query_7_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_13 = linear(bias = decoder_layers_1_encoder_attn_k_proj_bias, weight = decoder_layers_1_encoder_attn_k_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_15")]; + tensor value_states_13 = linear(bias = decoder_layers_1_encoder_attn_v_proj_bias, weight = decoder_layers_1_encoder_attn_v_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_16")]; + tensor concat_7x = const()[name = tensor("concat_7x"), val = tensor([1, -1, 16, 64])]; + tensor var_382 = reshape(shape = concat_7x, x = key_states_13)[name = tensor("op_382")]; + tensor key_states_15_perm_0 = const()[name = tensor("key_states_15_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_8x = const()[name = tensor("concat_8x"), val = tensor([1, -1, 16, 64])]; + tensor var_385 = reshape(shape = concat_8x, x = value_states_13)[name = tensor("op_385")]; + tensor value_states_15_perm_0 = const()[name = tensor("value_states_15_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_7_interleave_0 = const()[name = tensor("key_7_interleave_0"), val = tensor(false)]; + tensor key_states_15 = transpose(perm = key_states_15_perm_0, x = var_382)[name = tensor("transpose_203")]; + tensor key_7 = concat(axis = var_13, interleave = key_7_interleave_0, values = key_states_15)[name = tensor("key_7")]; + tensor value_7_interleave_0 = const()[name = tensor("value_7_interleave_0"), val = tensor(false)]; + tensor value_states_15 = transpose(perm = value_states_15_perm_0, x = var_385)[name = tensor("transpose_202")]; + tensor value_7 = concat(axis = var_13, interleave = value_7_interleave_0, values = value_states_15)[name = tensor("value_7")]; + tensor var_395_shape = shape(x = key_7)[name = tensor("op_395_shape")]; + tensor gather_5_indices_0 = const()[name = tensor("gather_5_indices_0"), val = tensor(2)]; + tensor gather_5_axis_0 = const()[name = tensor("gather_5_axis_0"), val = tensor(0)]; + tensor gather_5_batch_dims_0 = const()[name = tensor("gather_5_batch_dims_0"), val = tensor(0)]; + tensor gather_5 = gather(axis = gather_5_axis_0, batch_dims = gather_5_batch_dims_0, indices = gather_5_indices_0, x = var_395_shape)[name = tensor("gather_5")]; + tensor concat_9_values0_0 = const()[name = tensor("concat_9_values0_0"), val = tensor(0)]; + tensor concat_9_values1_0 = const()[name = tensor("concat_9_values1_0"), val = tensor(0)]; + tensor concat_9_values2_0 = const()[name = tensor("concat_9_values2_0"), val = tensor(0)]; + tensor concat_9_axis_0 = const()[name = tensor("concat_9_axis_0"), val = tensor(0)]; + tensor concat_9_interleave_0 = const()[name = tensor("concat_9_interleave_0"), val = tensor(false)]; + tensor concat_9 = concat(axis = concat_9_axis_0, interleave = concat_9_interleave_0, values = (concat_9_values0_0, concat_9_values1_0, concat_9_values2_0, gather_5))[name = tensor("concat_9")]; + tensor attention_mask_11_begin_0 = const()[name = tensor("attention_mask_11_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_11_end_mask_0 = const()[name = tensor("attention_mask_11_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_11 = slice_by_index(begin = attention_mask_11_begin_0, end = concat_9, end_mask = attention_mask_11_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_11")]; + tensor query_7 = transpose(perm = query_7_perm_0, x = var_373)[name = tensor("transpose_201")]; + tensor mul_3 = mul(x = query_7, y = var_11)[name = tensor("mul_3")]; + tensor matmul_3_transpose_y_0 = const()[name = tensor("matmul_3_transpose_y_0"), val = tensor(true)]; + tensor matmul_3_transpose_x_0 = const()[name = tensor("matmul_3_transpose_x_0"), val = tensor(false)]; + tensor matmul_3 = matmul(transpose_x = matmul_3_transpose_x_0, transpose_y = matmul_3_transpose_y_0, x = mul_3, y = key_7)[name = tensor("matmul_3")]; + tensor add_3 = add(x = matmul_3, y = attention_mask_11)[name = tensor("add_3")]; + tensor softmax_3_axis_0 = const()[name = tensor("softmax_3_axis_0"), val = tensor(-1)]; + tensor softmax_3 = softmax(axis = softmax_3_axis_0, x = add_3)[name = tensor("softmax_3")]; + tensor attn_output_13_transpose_x_0 = const()[name = tensor("attn_output_13_transpose_x_0"), val = tensor(false)]; + tensor attn_output_13_transpose_y_0 = const()[name = tensor("attn_output_13_transpose_y_0"), val = tensor(false)]; + tensor attn_output_13 = matmul(transpose_x = attn_output_13_transpose_x_0, transpose_y = attn_output_13_transpose_y_0, x = softmax_3, y = value_7)[name = tensor("attn_output_13")]; + tensor var_401_perm_0 = const()[name = tensor("op_401_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_403 = const()[name = tensor("op_403"), val = tensor([1, 2, -1])]; + tensor var_401 = transpose(perm = var_401_perm_0, x = attn_output_13)[name = tensor("transpose_200")]; + tensor var_404 = reshape(shape = var_403, x = var_401)[name = tensor("op_404")]; + tensor input_39 = linear(bias = decoder_layers_1_encoder_attn_out_proj_bias, weight = decoder_layers_1_encoder_attn_out_proj_weight_palettized, x = var_404)[name = tensor("linear_17")]; + tensor input_41 = add(x = input_35, y = input_39)[name = tensor("input_41")]; + tensor input_43_axes_0 = const()[name = tensor("input_43_axes_0"), val = tensor([-1])]; + tensor input_43 = layer_norm(axes = input_43_axes_0, beta = decoder_layers_1_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_1_final_layer_norm_weight, x = input_41)[name = tensor("input_43")]; + tensor input_45 = linear(bias = decoder_layers_1_fc1_bias, weight = decoder_layers_1_fc1_weight_palettized, x = input_43)[name = tensor("linear_18")]; + tensor input_47 = relu(x = input_45)[name = tensor("input_47")]; + tensor input_51 = linear(bias = decoder_layers_1_fc2_bias, weight = decoder_layers_1_fc2_weight_palettized, x = input_47)[name = tensor("linear_19")]; + tensor input_53 = add(x = input_41, y = input_51)[name = tensor("input_53")]; + tensor hidden_states_21_axes_0 = const()[name = tensor("hidden_states_21_axes_0"), val = tensor([-1])]; + tensor hidden_states_21 = layer_norm(axes = hidden_states_21_axes_0, beta = decoder_layers_2_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_2_self_attn_layer_norm_weight, x = input_53)[name = tensor("hidden_states_21")]; + tensor var_454 = linear(bias = decoder_layers_2_self_attn_q_proj_bias, weight = decoder_layers_2_self_attn_q_proj_weight_palettized, x = hidden_states_21)[name = tensor("linear_20")]; + tensor var_455 = const()[name = tensor("op_455"), val = tensor([1, 2, -1, 64])]; + tensor var_456 = reshape(shape = var_455, x = var_454)[name = tensor("op_456")]; + tensor key_states_17 = linear(bias = decoder_layers_2_self_attn_k_proj_bias, weight = decoder_layers_2_self_attn_k_proj_weight_palettized, x = hidden_states_21)[name = tensor("linear_21")]; + tensor value_states_17 = linear(bias = decoder_layers_2_self_attn_v_proj_bias, weight = decoder_layers_2_self_attn_v_proj_weight_palettized, x = hidden_states_21)[name = tensor("linear_22")]; + tensor var_464 = const()[name = tensor("op_464"), val = tensor([1, 2, -1, 64])]; + tensor var_465 = reshape(shape = var_464, x = key_states_17)[name = tensor("op_465")]; + tensor var_467 = const()[name = tensor("op_467"), val = tensor([1, 2, -1, 64])]; + tensor var_468 = reshape(shape = var_467, x = value_states_17)[name = tensor("op_468")]; + tensor value_states_19_perm_0 = const()[name = tensor("value_states_19_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_9_interleave_0 = const()[name = tensor("key_9_interleave_0"), val = tensor(false)]; + tensor const_125 = const()[name = tensor("const_125"), val = tensor(1)]; + tensor key_9 = concat(axis = const_125, interleave = key_9_interleave_0, values = var_465)[name = tensor("key_9")]; + tensor value_9_interleave_0 = const()[name = tensor("value_9_interleave_0"), val = tensor(false)]; + tensor value_states_19 = transpose(perm = value_states_19_perm_0, x = var_468)[name = tensor("transpose_199")]; + tensor value_9 = concat(axis = var_13, interleave = value_9_interleave_0, values = value_states_19)[name = tensor("value_9")]; + tensor mul_4 = mul(x = var_456, y = var_11)[name = tensor("mul_4")]; + tensor matmul_4_transpose_y_0 = const()[name = tensor("matmul_4_transpose_y_0"), val = tensor(true)]; + tensor matmul_4_transpose_x_0 = const()[name = tensor("matmul_4_transpose_x_0"), val = tensor(false)]; + tensor transpose_76_perm_0 = const()[name = tensor("transpose_76_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_77_perm_0 = const()[name = tensor("transpose_77_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_77 = transpose(perm = transpose_77_perm_0, x = key_9)[name = tensor("transpose_197")]; + tensor transpose_76 = transpose(perm = transpose_76_perm_0, x = mul_4)[name = tensor("transpose_198")]; + tensor matmul_4 = matmul(transpose_x = matmul_4_transpose_x_0, transpose_y = matmul_4_transpose_y_0, x = transpose_76, y = transpose_77)[name = tensor("matmul_4")]; + tensor add_4 = add(x = matmul_4, y = reshape_4)[name = tensor("add_4")]; + tensor softmax_4_axis_0 = const()[name = tensor("softmax_4_axis_0"), val = tensor(-1)]; + tensor softmax_4 = softmax(axis = softmax_4_axis_0, x = add_4)[name = tensor("softmax_4")]; + tensor attn_output_17_transpose_x_0 = const()[name = tensor("attn_output_17_transpose_x_0"), val = tensor(false)]; + tensor attn_output_17_transpose_y_0 = const()[name = tensor("attn_output_17_transpose_y_0"), val = tensor(false)]; + tensor attn_output_17 = matmul(transpose_x = attn_output_17_transpose_x_0, transpose_y = attn_output_17_transpose_y_0, x = softmax_4, y = value_9)[name = tensor("attn_output_17")]; + tensor var_484_perm_0 = const()[name = tensor("op_484_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_486 = const()[name = tensor("op_486"), val = tensor([1, 2, -1])]; + tensor var_484 = transpose(perm = var_484_perm_0, x = attn_output_17)[name = tensor("transpose_196")]; + tensor var_487 = reshape(shape = var_486, x = var_484)[name = tensor("op_487")]; + tensor input_57 = linear(bias = decoder_layers_2_self_attn_out_proj_bias, weight = decoder_layers_2_self_attn_out_proj_weight_palettized, x = var_487)[name = tensor("linear_23")]; + tensor input_59 = add(x = input_53, y = input_57)[name = tensor("input_59")]; + tensor hidden_states_25_axes_0 = const()[name = tensor("hidden_states_25_axes_0"), val = tensor([-1])]; + tensor hidden_states_25 = layer_norm(axes = hidden_states_25_axes_0, beta = decoder_layers_2_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_2_encoder_attn_layer_norm_weight, x = input_59)[name = tensor("hidden_states_25")]; + tensor var_511 = linear(bias = decoder_layers_2_encoder_attn_q_proj_bias, weight = decoder_layers_2_encoder_attn_q_proj_weight_palettized, x = hidden_states_25)[name = tensor("linear_24")]; + tensor var_512 = const()[name = tensor("op_512"), val = tensor([1, 2, -1, 64])]; + tensor var_513 = reshape(shape = var_512, x = var_511)[name = tensor("op_513")]; + tensor query_11_perm_0 = const()[name = tensor("query_11_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_21 = linear(bias = decoder_layers_2_encoder_attn_k_proj_bias, weight = decoder_layers_2_encoder_attn_k_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_25")]; + tensor value_states_21 = linear(bias = decoder_layers_2_encoder_attn_v_proj_bias, weight = decoder_layers_2_encoder_attn_v_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_26")]; + tensor concat_10x = const()[name = tensor("concat_10x"), val = tensor([1, -1, 16, 64])]; + tensor var_522 = reshape(shape = concat_10x, x = key_states_21)[name = tensor("op_522")]; + tensor key_states_23_perm_0 = const()[name = tensor("key_states_23_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_11x = const()[name = tensor("concat_11x"), val = tensor([1, -1, 16, 64])]; + tensor var_525 = reshape(shape = concat_11x, x = value_states_21)[name = tensor("op_525")]; + tensor value_states_23_perm_0 = const()[name = tensor("value_states_23_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_11_interleave_0 = const()[name = tensor("key_11_interleave_0"), val = tensor(false)]; + tensor key_states_23 = transpose(perm = key_states_23_perm_0, x = var_522)[name = tensor("transpose_195")]; + tensor key_11 = concat(axis = var_13, interleave = key_11_interleave_0, values = key_states_23)[name = tensor("key_11")]; + tensor value_11_interleave_0 = const()[name = tensor("value_11_interleave_0"), val = tensor(false)]; + tensor value_states_23 = transpose(perm = value_states_23_perm_0, x = var_525)[name = tensor("transpose_194")]; + tensor value_11 = concat(axis = var_13, interleave = value_11_interleave_0, values = value_states_23)[name = tensor("value_11")]; + tensor var_535_shape = shape(x = key_11)[name = tensor("op_535_shape")]; + tensor gather_7_indices_0 = const()[name = tensor("gather_7_indices_0"), val = tensor(2)]; + tensor gather_7_axis_0 = const()[name = tensor("gather_7_axis_0"), val = tensor(0)]; + tensor gather_7_batch_dims_0 = const()[name = tensor("gather_7_batch_dims_0"), val = tensor(0)]; + tensor gather_7 = gather(axis = gather_7_axis_0, batch_dims = gather_7_batch_dims_0, indices = gather_7_indices_0, x = var_535_shape)[name = tensor("gather_7")]; + tensor concat_12_values0_0 = const()[name = tensor("concat_12_values0_0"), val = tensor(0)]; + tensor concat_12_values1_0 = const()[name = tensor("concat_12_values1_0"), val = tensor(0)]; + tensor concat_12_values2_0 = const()[name = tensor("concat_12_values2_0"), val = tensor(0)]; + tensor concat_12_axis_0 = const()[name = tensor("concat_12_axis_0"), val = tensor(0)]; + tensor concat_12_interleave_0 = const()[name = tensor("concat_12_interleave_0"), val = tensor(false)]; + tensor concat_12 = concat(axis = concat_12_axis_0, interleave = concat_12_interleave_0, values = (concat_12_values0_0, concat_12_values1_0, concat_12_values2_0, gather_7))[name = tensor("concat_12")]; + tensor attention_mask_15_begin_0 = const()[name = tensor("attention_mask_15_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_15_end_mask_0 = const()[name = tensor("attention_mask_15_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_15 = slice_by_index(begin = attention_mask_15_begin_0, end = concat_12, end_mask = attention_mask_15_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_15")]; + tensor query_11 = transpose(perm = query_11_perm_0, x = var_513)[name = tensor("transpose_193")]; + tensor mul_5 = mul(x = query_11, y = var_11)[name = tensor("mul_5")]; + tensor matmul_5_transpose_y_0 = const()[name = tensor("matmul_5_transpose_y_0"), val = tensor(true)]; + tensor matmul_5_transpose_x_0 = const()[name = tensor("matmul_5_transpose_x_0"), val = tensor(false)]; + tensor matmul_5 = matmul(transpose_x = matmul_5_transpose_x_0, transpose_y = matmul_5_transpose_y_0, x = mul_5, y = key_11)[name = tensor("matmul_5")]; + tensor add_5 = add(x = matmul_5, y = attention_mask_15)[name = tensor("add_5")]; + tensor softmax_5_axis_0 = const()[name = tensor("softmax_5_axis_0"), val = tensor(-1)]; + tensor softmax_5 = softmax(axis = softmax_5_axis_0, x = add_5)[name = tensor("softmax_5")]; + tensor attn_output_21_transpose_x_0 = const()[name = tensor("attn_output_21_transpose_x_0"), val = tensor(false)]; + tensor attn_output_21_transpose_y_0 = const()[name = tensor("attn_output_21_transpose_y_0"), val = tensor(false)]; + tensor attn_output_21 = matmul(transpose_x = attn_output_21_transpose_x_0, transpose_y = attn_output_21_transpose_y_0, x = softmax_5, y = value_11)[name = tensor("attn_output_21")]; + tensor var_541_perm_0 = const()[name = tensor("op_541_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_543 = const()[name = tensor("op_543"), val = tensor([1, 2, -1])]; + tensor var_541 = transpose(perm = var_541_perm_0, x = attn_output_21)[name = tensor("transpose_192")]; + tensor var_544 = reshape(shape = var_543, x = var_541)[name = tensor("op_544")]; + tensor input_63 = linear(bias = decoder_layers_2_encoder_attn_out_proj_bias, weight = decoder_layers_2_encoder_attn_out_proj_weight_palettized, x = var_544)[name = tensor("linear_27")]; + tensor input_65 = add(x = input_59, y = input_63)[name = tensor("input_65")]; + tensor input_67_axes_0 = const()[name = tensor("input_67_axes_0"), val = tensor([-1])]; + tensor input_67 = layer_norm(axes = input_67_axes_0, beta = decoder_layers_2_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_2_final_layer_norm_weight, x = input_65)[name = tensor("input_67")]; + tensor input_69 = linear(bias = decoder_layers_2_fc1_bias, weight = decoder_layers_2_fc1_weight_palettized, x = input_67)[name = tensor("linear_28")]; + tensor input_71 = relu(x = input_69)[name = tensor("input_71")]; + tensor input_75 = linear(bias = decoder_layers_2_fc2_bias, weight = decoder_layers_2_fc2_weight_palettized, x = input_71)[name = tensor("linear_29")]; + tensor input_77 = add(x = input_65, y = input_75)[name = tensor("input_77")]; + tensor hidden_states_31_axes_0 = const()[name = tensor("hidden_states_31_axes_0"), val = tensor([-1])]; + tensor hidden_states_31 = layer_norm(axes = hidden_states_31_axes_0, beta = decoder_layers_3_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_3_self_attn_layer_norm_weight, x = input_77)[name = tensor("hidden_states_31")]; + tensor var_594 = linear(bias = decoder_layers_3_self_attn_q_proj_bias, weight = decoder_layers_3_self_attn_q_proj_weight_palettized, x = hidden_states_31)[name = tensor("linear_30")]; + tensor var_595 = const()[name = tensor("op_595"), val = tensor([1, 2, -1, 64])]; + tensor var_596 = reshape(shape = var_595, x = var_594)[name = tensor("op_596")]; + tensor key_states_25 = linear(bias = decoder_layers_3_self_attn_k_proj_bias, weight = decoder_layers_3_self_attn_k_proj_weight_palettized, x = hidden_states_31)[name = tensor("linear_31")]; + tensor value_states_25 = linear(bias = decoder_layers_3_self_attn_v_proj_bias, weight = decoder_layers_3_self_attn_v_proj_weight_palettized, x = hidden_states_31)[name = tensor("linear_32")]; + tensor var_604 = const()[name = tensor("op_604"), val = tensor([1, 2, -1, 64])]; + tensor var_605 = reshape(shape = var_604, x = key_states_25)[name = tensor("op_605")]; + tensor var_607 = const()[name = tensor("op_607"), val = tensor([1, 2, -1, 64])]; + tensor var_608 = reshape(shape = var_607, x = value_states_25)[name = tensor("op_608")]; + tensor value_states_27_perm_0 = const()[name = tensor("value_states_27_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_13_interleave_0 = const()[name = tensor("key_13_interleave_0"), val = tensor(false)]; + tensor const_126 = const()[name = tensor("const_126"), val = tensor(1)]; + tensor key_13 = concat(axis = const_126, interleave = key_13_interleave_0, values = var_605)[name = tensor("key_13")]; + tensor value_13_interleave_0 = const()[name = tensor("value_13_interleave_0"), val = tensor(false)]; + tensor value_states_27 = transpose(perm = value_states_27_perm_0, x = var_608)[name = tensor("transpose_191")]; + tensor value_13 = concat(axis = var_13, interleave = value_13_interleave_0, values = value_states_27)[name = tensor("value_13")]; + tensor mul_6 = mul(x = var_596, y = var_11)[name = tensor("mul_6")]; + tensor matmul_6_transpose_y_0 = const()[name = tensor("matmul_6_transpose_y_0"), val = tensor(true)]; + tensor matmul_6_transpose_x_0 = const()[name = tensor("matmul_6_transpose_x_0"), val = tensor(false)]; + tensor transpose_78_perm_0 = const()[name = tensor("transpose_78_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_79_perm_0 = const()[name = tensor("transpose_79_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_79 = transpose(perm = transpose_79_perm_0, x = key_13)[name = tensor("transpose_189")]; + tensor transpose_78 = transpose(perm = transpose_78_perm_0, x = mul_6)[name = tensor("transpose_190")]; + tensor matmul_6 = matmul(transpose_x = matmul_6_transpose_x_0, transpose_y = matmul_6_transpose_y_0, x = transpose_78, y = transpose_79)[name = tensor("matmul_6")]; + tensor add_6 = add(x = matmul_6, y = reshape_4)[name = tensor("add_6")]; + tensor softmax_6_axis_0 = const()[name = tensor("softmax_6_axis_0"), val = tensor(-1)]; + tensor softmax_6 = softmax(axis = softmax_6_axis_0, x = add_6)[name = tensor("softmax_6")]; + tensor attn_output_25_transpose_x_0 = const()[name = tensor("attn_output_25_transpose_x_0"), val = tensor(false)]; + tensor attn_output_25_transpose_y_0 = const()[name = tensor("attn_output_25_transpose_y_0"), val = tensor(false)]; + tensor attn_output_25 = matmul(transpose_x = attn_output_25_transpose_x_0, transpose_y = attn_output_25_transpose_y_0, x = softmax_6, y = value_13)[name = tensor("attn_output_25")]; + tensor var_624_perm_0 = const()[name = tensor("op_624_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_626 = const()[name = tensor("op_626"), val = tensor([1, 2, -1])]; + tensor var_624 = transpose(perm = var_624_perm_0, x = attn_output_25)[name = tensor("transpose_188")]; + tensor var_627 = reshape(shape = var_626, x = var_624)[name = tensor("op_627")]; + tensor input_81 = linear(bias = decoder_layers_3_self_attn_out_proj_bias, weight = decoder_layers_3_self_attn_out_proj_weight_palettized, x = var_627)[name = tensor("linear_33")]; + tensor input_83 = add(x = input_77, y = input_81)[name = tensor("input_83")]; + tensor hidden_states_35_axes_0 = const()[name = tensor("hidden_states_35_axes_0"), val = tensor([-1])]; + tensor hidden_states_35 = layer_norm(axes = hidden_states_35_axes_0, beta = decoder_layers_3_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_3_encoder_attn_layer_norm_weight, x = input_83)[name = tensor("hidden_states_35")]; + tensor var_651 = linear(bias = decoder_layers_3_encoder_attn_q_proj_bias, weight = decoder_layers_3_encoder_attn_q_proj_weight_palettized, x = hidden_states_35)[name = tensor("linear_34")]; + tensor var_652 = const()[name = tensor("op_652"), val = tensor([1, 2, -1, 64])]; + tensor var_653 = reshape(shape = var_652, x = var_651)[name = tensor("op_653")]; + tensor query_15_perm_0 = const()[name = tensor("query_15_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_29 = linear(bias = decoder_layers_3_encoder_attn_k_proj_bias, weight = decoder_layers_3_encoder_attn_k_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_35")]; + tensor value_states_29 = linear(bias = decoder_layers_3_encoder_attn_v_proj_bias, weight = decoder_layers_3_encoder_attn_v_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_36")]; + tensor concat_13x = const()[name = tensor("concat_13x"), val = tensor([1, -1, 16, 64])]; + tensor var_662 = reshape(shape = concat_13x, x = key_states_29)[name = tensor("op_662")]; + tensor key_states_31_perm_0 = const()[name = tensor("key_states_31_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_14x = const()[name = tensor("concat_14x"), val = tensor([1, -1, 16, 64])]; + tensor var_665 = reshape(shape = concat_14x, x = value_states_29)[name = tensor("op_665")]; + tensor value_states_31_perm_0 = const()[name = tensor("value_states_31_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_15_interleave_0 = const()[name = tensor("key_15_interleave_0"), val = tensor(false)]; + tensor key_states_31 = transpose(perm = key_states_31_perm_0, x = var_662)[name = tensor("transpose_187")]; + tensor key_15 = concat(axis = var_13, interleave = key_15_interleave_0, values = key_states_31)[name = tensor("key_15")]; + tensor value_15_interleave_0 = const()[name = tensor("value_15_interleave_0"), val = tensor(false)]; + tensor value_states_31 = transpose(perm = value_states_31_perm_0, x = var_665)[name = tensor("transpose_186")]; + tensor value_15 = concat(axis = var_13, interleave = value_15_interleave_0, values = value_states_31)[name = tensor("value_15")]; + tensor var_675_shape = shape(x = key_15)[name = tensor("op_675_shape")]; + tensor gather_9_indices_0 = const()[name = tensor("gather_9_indices_0"), val = tensor(2)]; + tensor gather_9_axis_0 = const()[name = tensor("gather_9_axis_0"), val = tensor(0)]; + tensor gather_9_batch_dims_0 = const()[name = tensor("gather_9_batch_dims_0"), val = tensor(0)]; + tensor gather_9 = gather(axis = gather_9_axis_0, batch_dims = gather_9_batch_dims_0, indices = gather_9_indices_0, x = var_675_shape)[name = tensor("gather_9")]; + tensor concat_15_values0_0 = const()[name = tensor("concat_15_values0_0"), val = tensor(0)]; + tensor concat_15_values1_0 = const()[name = tensor("concat_15_values1_0"), val = tensor(0)]; + tensor concat_15_values2_0 = const()[name = tensor("concat_15_values2_0"), val = tensor(0)]; + tensor concat_15_axis_0 = const()[name = tensor("concat_15_axis_0"), val = tensor(0)]; + tensor concat_15_interleave_0 = const()[name = tensor("concat_15_interleave_0"), val = tensor(false)]; + tensor concat_15 = concat(axis = concat_15_axis_0, interleave = concat_15_interleave_0, values = (concat_15_values0_0, concat_15_values1_0, concat_15_values2_0, gather_9))[name = tensor("concat_15")]; + tensor attention_mask_19_begin_0 = const()[name = tensor("attention_mask_19_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_19_end_mask_0 = const()[name = tensor("attention_mask_19_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_19 = slice_by_index(begin = attention_mask_19_begin_0, end = concat_15, end_mask = attention_mask_19_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_19")]; + tensor query_15 = transpose(perm = query_15_perm_0, x = var_653)[name = tensor("transpose_185")]; + tensor mul_7 = mul(x = query_15, y = var_11)[name = tensor("mul_7")]; + tensor matmul_7_transpose_y_0 = const()[name = tensor("matmul_7_transpose_y_0"), val = tensor(true)]; + tensor matmul_7_transpose_x_0 = const()[name = tensor("matmul_7_transpose_x_0"), val = tensor(false)]; + tensor matmul_7 = matmul(transpose_x = matmul_7_transpose_x_0, transpose_y = matmul_7_transpose_y_0, x = mul_7, y = key_15)[name = tensor("matmul_7")]; + tensor add_7 = add(x = matmul_7, y = attention_mask_19)[name = tensor("add_7")]; + tensor softmax_7_axis_0 = const()[name = tensor("softmax_7_axis_0"), val = tensor(-1)]; + tensor softmax_7 = softmax(axis = softmax_7_axis_0, x = add_7)[name = tensor("softmax_7")]; + tensor attn_output_29_transpose_x_0 = const()[name = tensor("attn_output_29_transpose_x_0"), val = tensor(false)]; + tensor attn_output_29_transpose_y_0 = const()[name = tensor("attn_output_29_transpose_y_0"), val = tensor(false)]; + tensor attn_output_29 = matmul(transpose_x = attn_output_29_transpose_x_0, transpose_y = attn_output_29_transpose_y_0, x = softmax_7, y = value_15)[name = tensor("attn_output_29")]; + tensor var_681_perm_0 = const()[name = tensor("op_681_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_683 = const()[name = tensor("op_683"), val = tensor([1, 2, -1])]; + tensor var_681 = transpose(perm = var_681_perm_0, x = attn_output_29)[name = tensor("transpose_184")]; + tensor var_684 = reshape(shape = var_683, x = var_681)[name = tensor("op_684")]; + tensor input_87 = linear(bias = decoder_layers_3_encoder_attn_out_proj_bias, weight = decoder_layers_3_encoder_attn_out_proj_weight_palettized, x = var_684)[name = tensor("linear_37")]; + tensor input_89 = add(x = input_83, y = input_87)[name = tensor("input_89")]; + tensor input_91_axes_0 = const()[name = tensor("input_91_axes_0"), val = tensor([-1])]; + tensor input_91 = layer_norm(axes = input_91_axes_0, beta = decoder_layers_3_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_3_final_layer_norm_weight, x = input_89)[name = tensor("input_91")]; + tensor input_93 = linear(bias = decoder_layers_3_fc1_bias, weight = decoder_layers_3_fc1_weight_palettized, x = input_91)[name = tensor("linear_38")]; + tensor input_95 = relu(x = input_93)[name = tensor("input_95")]; + tensor input_99 = linear(bias = decoder_layers_3_fc2_bias, weight = decoder_layers_3_fc2_weight_palettized, x = input_95)[name = tensor("linear_39")]; + tensor input_101 = add(x = input_89, y = input_99)[name = tensor("input_101")]; + tensor hidden_states_41_axes_0 = const()[name = tensor("hidden_states_41_axes_0"), val = tensor([-1])]; + tensor hidden_states_41 = layer_norm(axes = hidden_states_41_axes_0, beta = decoder_layers_4_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_4_self_attn_layer_norm_weight, x = input_101)[name = tensor("hidden_states_41")]; + tensor var_734 = linear(bias = decoder_layers_4_self_attn_q_proj_bias, weight = decoder_layers_4_self_attn_q_proj_weight_palettized, x = hidden_states_41)[name = tensor("linear_40")]; + tensor var_735 = const()[name = tensor("op_735"), val = tensor([1, 2, -1, 64])]; + tensor var_736 = reshape(shape = var_735, x = var_734)[name = tensor("op_736")]; + tensor key_states_33 = linear(bias = decoder_layers_4_self_attn_k_proj_bias, weight = decoder_layers_4_self_attn_k_proj_weight_palettized, x = hidden_states_41)[name = tensor("linear_41")]; + tensor value_states_33 = linear(bias = decoder_layers_4_self_attn_v_proj_bias, weight = decoder_layers_4_self_attn_v_proj_weight_palettized, x = hidden_states_41)[name = tensor("linear_42")]; + tensor var_744 = const()[name = tensor("op_744"), val = tensor([1, 2, -1, 64])]; + tensor var_745 = reshape(shape = var_744, x = key_states_33)[name = tensor("op_745")]; + tensor var_747 = const()[name = tensor("op_747"), val = tensor([1, 2, -1, 64])]; + tensor var_748 = reshape(shape = var_747, x = value_states_33)[name = tensor("op_748")]; + tensor value_states_35_perm_0 = const()[name = tensor("value_states_35_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_17_interleave_0 = const()[name = tensor("key_17_interleave_0"), val = tensor(false)]; + tensor const_127 = const()[name = tensor("const_127"), val = tensor(1)]; + tensor key_17 = concat(axis = const_127, interleave = key_17_interleave_0, values = var_745)[name = tensor("key_17")]; + tensor value_17_interleave_0 = const()[name = tensor("value_17_interleave_0"), val = tensor(false)]; + tensor value_states_35 = transpose(perm = value_states_35_perm_0, x = var_748)[name = tensor("transpose_183")]; + tensor value_17 = concat(axis = var_13, interleave = value_17_interleave_0, values = value_states_35)[name = tensor("value_17")]; + tensor mul_8 = mul(x = var_736, y = var_11)[name = tensor("mul_8")]; + tensor matmul_8_transpose_y_0 = const()[name = tensor("matmul_8_transpose_y_0"), val = tensor(true)]; + tensor matmul_8_transpose_x_0 = const()[name = tensor("matmul_8_transpose_x_0"), val = tensor(false)]; + tensor transpose_80_perm_0 = const()[name = tensor("transpose_80_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_81_perm_0 = const()[name = tensor("transpose_81_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_81 = transpose(perm = transpose_81_perm_0, x = key_17)[name = tensor("transpose_181")]; + tensor transpose_80 = transpose(perm = transpose_80_perm_0, x = mul_8)[name = tensor("transpose_182")]; + tensor matmul_8 = matmul(transpose_x = matmul_8_transpose_x_0, transpose_y = matmul_8_transpose_y_0, x = transpose_80, y = transpose_81)[name = tensor("matmul_8")]; + tensor add_8 = add(x = matmul_8, y = reshape_4)[name = tensor("add_8")]; + tensor softmax_8_axis_0 = const()[name = tensor("softmax_8_axis_0"), val = tensor(-1)]; + tensor softmax_8 = softmax(axis = softmax_8_axis_0, x = add_8)[name = tensor("softmax_8")]; + tensor attn_output_33_transpose_x_0 = const()[name = tensor("attn_output_33_transpose_x_0"), val = tensor(false)]; + tensor attn_output_33_transpose_y_0 = const()[name = tensor("attn_output_33_transpose_y_0"), val = tensor(false)]; + tensor attn_output_33 = matmul(transpose_x = attn_output_33_transpose_x_0, transpose_y = attn_output_33_transpose_y_0, x = softmax_8, y = value_17)[name = tensor("attn_output_33")]; + tensor var_764_perm_0 = const()[name = tensor("op_764_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_766 = const()[name = tensor("op_766"), val = tensor([1, 2, -1])]; + tensor var_764 = transpose(perm = var_764_perm_0, x = attn_output_33)[name = tensor("transpose_180")]; + tensor var_767 = reshape(shape = var_766, x = var_764)[name = tensor("op_767")]; + tensor input_105 = linear(bias = decoder_layers_4_self_attn_out_proj_bias, weight = decoder_layers_4_self_attn_out_proj_weight_palettized, x = var_767)[name = tensor("linear_43")]; + tensor input_107 = add(x = input_101, y = input_105)[name = tensor("input_107")]; + tensor hidden_states_45_axes_0 = const()[name = tensor("hidden_states_45_axes_0"), val = tensor([-1])]; + tensor hidden_states_45 = layer_norm(axes = hidden_states_45_axes_0, beta = decoder_layers_4_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_4_encoder_attn_layer_norm_weight, x = input_107)[name = tensor("hidden_states_45")]; + tensor var_791 = linear(bias = decoder_layers_4_encoder_attn_q_proj_bias, weight = decoder_layers_4_encoder_attn_q_proj_weight_palettized, x = hidden_states_45)[name = tensor("linear_44")]; + tensor var_792 = const()[name = tensor("op_792"), val = tensor([1, 2, -1, 64])]; + tensor var_793 = reshape(shape = var_792, x = var_791)[name = tensor("op_793")]; + tensor query_19_perm_0 = const()[name = tensor("query_19_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_37 = linear(bias = decoder_layers_4_encoder_attn_k_proj_bias, weight = decoder_layers_4_encoder_attn_k_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_45")]; + tensor value_states_37 = linear(bias = decoder_layers_4_encoder_attn_v_proj_bias, weight = decoder_layers_4_encoder_attn_v_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_46")]; + tensor concat_16x = const()[name = tensor("concat_16x"), val = tensor([1, -1, 16, 64])]; + tensor var_802 = reshape(shape = concat_16x, x = key_states_37)[name = tensor("op_802")]; + tensor key_states_39_perm_0 = const()[name = tensor("key_states_39_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_17x = const()[name = tensor("concat_17x"), val = tensor([1, -1, 16, 64])]; + tensor var_805 = reshape(shape = concat_17x, x = value_states_37)[name = tensor("op_805")]; + tensor value_states_39_perm_0 = const()[name = tensor("value_states_39_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_19_interleave_0 = const()[name = tensor("key_19_interleave_0"), val = tensor(false)]; + tensor key_states_39 = transpose(perm = key_states_39_perm_0, x = var_802)[name = tensor("transpose_179")]; + tensor key_19 = concat(axis = var_13, interleave = key_19_interleave_0, values = key_states_39)[name = tensor("key_19")]; + tensor value_19_interleave_0 = const()[name = tensor("value_19_interleave_0"), val = tensor(false)]; + tensor value_states_39 = transpose(perm = value_states_39_perm_0, x = var_805)[name = tensor("transpose_178")]; + tensor value_19 = concat(axis = var_13, interleave = value_19_interleave_0, values = value_states_39)[name = tensor("value_19")]; + tensor var_815_shape = shape(x = key_19)[name = tensor("op_815_shape")]; + tensor gather_11_indices_0 = const()[name = tensor("gather_11_indices_0"), val = tensor(2)]; + tensor gather_11_axis_0 = const()[name = tensor("gather_11_axis_0"), val = tensor(0)]; + tensor gather_11_batch_dims_0 = const()[name = tensor("gather_11_batch_dims_0"), val = tensor(0)]; + tensor gather_11 = gather(axis = gather_11_axis_0, batch_dims = gather_11_batch_dims_0, indices = gather_11_indices_0, x = var_815_shape)[name = tensor("gather_11")]; + tensor concat_18_values0_0 = const()[name = tensor("concat_18_values0_0"), val = tensor(0)]; + tensor concat_18_values1_0 = const()[name = tensor("concat_18_values1_0"), val = tensor(0)]; + tensor concat_18_values2_0 = const()[name = tensor("concat_18_values2_0"), val = tensor(0)]; + tensor concat_18_axis_0 = const()[name = tensor("concat_18_axis_0"), val = tensor(0)]; + tensor concat_18_interleave_0 = const()[name = tensor("concat_18_interleave_0"), val = tensor(false)]; + tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (concat_18_values0_0, concat_18_values1_0, concat_18_values2_0, gather_11))[name = tensor("concat_18")]; + tensor attention_mask_23_begin_0 = const()[name = tensor("attention_mask_23_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_23_end_mask_0 = const()[name = tensor("attention_mask_23_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_23 = slice_by_index(begin = attention_mask_23_begin_0, end = concat_18, end_mask = attention_mask_23_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_23")]; + tensor query_19 = transpose(perm = query_19_perm_0, x = var_793)[name = tensor("transpose_177")]; + tensor mul_9 = mul(x = query_19, y = var_11)[name = tensor("mul_9")]; + tensor matmul_9_transpose_y_0 = const()[name = tensor("matmul_9_transpose_y_0"), val = tensor(true)]; + tensor matmul_9_transpose_x_0 = const()[name = tensor("matmul_9_transpose_x_0"), val = tensor(false)]; + tensor matmul_9 = matmul(transpose_x = matmul_9_transpose_x_0, transpose_y = matmul_9_transpose_y_0, x = mul_9, y = key_19)[name = tensor("matmul_9")]; + tensor add_9 = add(x = matmul_9, y = attention_mask_23)[name = tensor("add_9")]; + tensor softmax_9_axis_0 = const()[name = tensor("softmax_9_axis_0"), val = tensor(-1)]; + tensor softmax_9 = softmax(axis = softmax_9_axis_0, x = add_9)[name = tensor("softmax_9")]; + tensor attn_output_37_transpose_x_0 = const()[name = tensor("attn_output_37_transpose_x_0"), val = tensor(false)]; + tensor attn_output_37_transpose_y_0 = const()[name = tensor("attn_output_37_transpose_y_0"), val = tensor(false)]; + tensor attn_output_37 = matmul(transpose_x = attn_output_37_transpose_x_0, transpose_y = attn_output_37_transpose_y_0, x = softmax_9, y = value_19)[name = tensor("attn_output_37")]; + tensor var_821_perm_0 = const()[name = tensor("op_821_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_823 = const()[name = tensor("op_823"), val = tensor([1, 2, -1])]; + tensor var_821 = transpose(perm = var_821_perm_0, x = attn_output_37)[name = tensor("transpose_176")]; + tensor var_824 = reshape(shape = var_823, x = var_821)[name = tensor("op_824")]; + tensor input_111 = linear(bias = decoder_layers_4_encoder_attn_out_proj_bias, weight = decoder_layers_4_encoder_attn_out_proj_weight_palettized, x = var_824)[name = tensor("linear_47")]; + tensor input_113 = add(x = input_107, y = input_111)[name = tensor("input_113")]; + tensor input_115_axes_0 = const()[name = tensor("input_115_axes_0"), val = tensor([-1])]; + tensor input_115 = layer_norm(axes = input_115_axes_0, beta = decoder_layers_4_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_4_final_layer_norm_weight, x = input_113)[name = tensor("input_115")]; + tensor input_117 = linear(bias = decoder_layers_4_fc1_bias, weight = decoder_layers_4_fc1_weight_palettized, x = input_115)[name = tensor("linear_48")]; + tensor input_119 = relu(x = input_117)[name = tensor("input_119")]; + tensor input_123 = linear(bias = decoder_layers_4_fc2_bias, weight = decoder_layers_4_fc2_weight_palettized, x = input_119)[name = tensor("linear_49")]; + tensor input_125 = add(x = input_113, y = input_123)[name = tensor("input_125")]; + tensor hidden_states_51_axes_0 = const()[name = tensor("hidden_states_51_axes_0"), val = tensor([-1])]; + tensor hidden_states_51 = layer_norm(axes = hidden_states_51_axes_0, beta = decoder_layers_5_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_5_self_attn_layer_norm_weight, x = input_125)[name = tensor("hidden_states_51")]; + tensor var_874 = linear(bias = decoder_layers_5_self_attn_q_proj_bias, weight = decoder_layers_5_self_attn_q_proj_weight_palettized, x = hidden_states_51)[name = tensor("linear_50")]; + tensor var_875 = const()[name = tensor("op_875"), val = tensor([1, 2, -1, 64])]; + tensor var_876 = reshape(shape = var_875, x = var_874)[name = tensor("op_876")]; + tensor key_states_41 = linear(bias = decoder_layers_5_self_attn_k_proj_bias, weight = decoder_layers_5_self_attn_k_proj_weight_palettized, x = hidden_states_51)[name = tensor("linear_51")]; + tensor value_states_41 = linear(bias = decoder_layers_5_self_attn_v_proj_bias, weight = decoder_layers_5_self_attn_v_proj_weight_palettized, x = hidden_states_51)[name = tensor("linear_52")]; + tensor var_884 = const()[name = tensor("op_884"), val = tensor([1, 2, -1, 64])]; + tensor var_885 = reshape(shape = var_884, x = key_states_41)[name = tensor("op_885")]; + tensor var_887 = const()[name = tensor("op_887"), val = tensor([1, 2, -1, 64])]; + tensor var_888 = reshape(shape = var_887, x = value_states_41)[name = tensor("op_888")]; + tensor value_states_43_perm_0 = const()[name = tensor("value_states_43_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_21_interleave_0 = const()[name = tensor("key_21_interleave_0"), val = tensor(false)]; + tensor const_128 = const()[name = tensor("const_128"), val = tensor(1)]; + tensor key_21 = concat(axis = const_128, interleave = key_21_interleave_0, values = var_885)[name = tensor("key_21")]; + tensor value_21_interleave_0 = const()[name = tensor("value_21_interleave_0"), val = tensor(false)]; + tensor value_states_43 = transpose(perm = value_states_43_perm_0, x = var_888)[name = tensor("transpose_175")]; + tensor value_21 = concat(axis = var_13, interleave = value_21_interleave_0, values = value_states_43)[name = tensor("value_21")]; + tensor mul_10 = mul(x = var_876, y = var_11)[name = tensor("mul_10")]; + tensor matmul_10_transpose_y_0 = const()[name = tensor("matmul_10_transpose_y_0"), val = tensor(true)]; + tensor matmul_10_transpose_x_0 = const()[name = tensor("matmul_10_transpose_x_0"), val = tensor(false)]; + tensor transpose_82_perm_0 = const()[name = tensor("transpose_82_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_83_perm_0 = const()[name = tensor("transpose_83_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_83 = transpose(perm = transpose_83_perm_0, x = key_21)[name = tensor("transpose_173")]; + tensor transpose_82 = transpose(perm = transpose_82_perm_0, x = mul_10)[name = tensor("transpose_174")]; + tensor matmul_10 = matmul(transpose_x = matmul_10_transpose_x_0, transpose_y = matmul_10_transpose_y_0, x = transpose_82, y = transpose_83)[name = tensor("matmul_10")]; + tensor add_10 = add(x = matmul_10, y = reshape_4)[name = tensor("add_10")]; + tensor softmax_10_axis_0 = const()[name = tensor("softmax_10_axis_0"), val = tensor(-1)]; + tensor softmax_10 = softmax(axis = softmax_10_axis_0, x = add_10)[name = tensor("softmax_10")]; + tensor attn_output_41_transpose_x_0 = const()[name = tensor("attn_output_41_transpose_x_0"), val = tensor(false)]; + tensor attn_output_41_transpose_y_0 = const()[name = tensor("attn_output_41_transpose_y_0"), val = tensor(false)]; + tensor attn_output_41 = matmul(transpose_x = attn_output_41_transpose_x_0, transpose_y = attn_output_41_transpose_y_0, x = softmax_10, y = value_21)[name = tensor("attn_output_41")]; + tensor var_904_perm_0 = const()[name = tensor("op_904_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_906 = const()[name = tensor("op_906"), val = tensor([1, 2, -1])]; + tensor var_904 = transpose(perm = var_904_perm_0, x = attn_output_41)[name = tensor("transpose_172")]; + tensor var_907 = reshape(shape = var_906, x = var_904)[name = tensor("op_907")]; + tensor input_129 = linear(bias = decoder_layers_5_self_attn_out_proj_bias, weight = decoder_layers_5_self_attn_out_proj_weight_palettized, x = var_907)[name = tensor("linear_53")]; + tensor input_131 = add(x = input_125, y = input_129)[name = tensor("input_131")]; + tensor hidden_states_55_axes_0 = const()[name = tensor("hidden_states_55_axes_0"), val = tensor([-1])]; + tensor hidden_states_55 = layer_norm(axes = hidden_states_55_axes_0, beta = decoder_layers_5_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_5_encoder_attn_layer_norm_weight, x = input_131)[name = tensor("hidden_states_55")]; + tensor var_931 = linear(bias = decoder_layers_5_encoder_attn_q_proj_bias, weight = decoder_layers_5_encoder_attn_q_proj_weight_palettized, x = hidden_states_55)[name = tensor("linear_54")]; + tensor var_932 = const()[name = tensor("op_932"), val = tensor([1, 2, -1, 64])]; + tensor var_933 = reshape(shape = var_932, x = var_931)[name = tensor("op_933")]; + tensor query_23_perm_0 = const()[name = tensor("query_23_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_45 = linear(bias = decoder_layers_5_encoder_attn_k_proj_bias, weight = decoder_layers_5_encoder_attn_k_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_55")]; + tensor value_states_45 = linear(bias = decoder_layers_5_encoder_attn_v_proj_bias, weight = decoder_layers_5_encoder_attn_v_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_56")]; + tensor concat_19x = const()[name = tensor("concat_19x"), val = tensor([1, -1, 16, 64])]; + tensor var_942 = reshape(shape = concat_19x, x = key_states_45)[name = tensor("op_942")]; + tensor key_states_47_perm_0 = const()[name = tensor("key_states_47_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_20x = const()[name = tensor("concat_20x"), val = tensor([1, -1, 16, 64])]; + tensor var_945 = reshape(shape = concat_20x, x = value_states_45)[name = tensor("op_945")]; + tensor value_states_47_perm_0 = const()[name = tensor("value_states_47_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_23_interleave_0 = const()[name = tensor("key_23_interleave_0"), val = tensor(false)]; + tensor key_states_47 = transpose(perm = key_states_47_perm_0, x = var_942)[name = tensor("transpose_171")]; + tensor key_23 = concat(axis = var_13, interleave = key_23_interleave_0, values = key_states_47)[name = tensor("key_23")]; + tensor value_23_interleave_0 = const()[name = tensor("value_23_interleave_0"), val = tensor(false)]; + tensor value_states_47 = transpose(perm = value_states_47_perm_0, x = var_945)[name = tensor("transpose_170")]; + tensor value_23 = concat(axis = var_13, interleave = value_23_interleave_0, values = value_states_47)[name = tensor("value_23")]; + tensor var_955_shape = shape(x = key_23)[name = tensor("op_955_shape")]; + tensor gather_13_indices_0 = const()[name = tensor("gather_13_indices_0"), val = tensor(2)]; + tensor gather_13_axis_0 = const()[name = tensor("gather_13_axis_0"), val = tensor(0)]; + tensor gather_13_batch_dims_0 = const()[name = tensor("gather_13_batch_dims_0"), val = tensor(0)]; + tensor gather_13 = gather(axis = gather_13_axis_0, batch_dims = gather_13_batch_dims_0, indices = gather_13_indices_0, x = var_955_shape)[name = tensor("gather_13")]; + tensor concat_21_values0_0 = const()[name = tensor("concat_21_values0_0"), val = tensor(0)]; + tensor concat_21_values1_0 = const()[name = tensor("concat_21_values1_0"), val = tensor(0)]; + tensor concat_21_values2_0 = const()[name = tensor("concat_21_values2_0"), val = tensor(0)]; + tensor concat_21_axis_0 = const()[name = tensor("concat_21_axis_0"), val = tensor(0)]; + tensor concat_21_interleave_0 = const()[name = tensor("concat_21_interleave_0"), val = tensor(false)]; + tensor concat_21 = concat(axis = concat_21_axis_0, interleave = concat_21_interleave_0, values = (concat_21_values0_0, concat_21_values1_0, concat_21_values2_0, gather_13))[name = tensor("concat_21")]; + tensor attention_mask_27_begin_0 = const()[name = tensor("attention_mask_27_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_27_end_mask_0 = const()[name = tensor("attention_mask_27_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_27 = slice_by_index(begin = attention_mask_27_begin_0, end = concat_21, end_mask = attention_mask_27_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_27")]; + tensor query_23 = transpose(perm = query_23_perm_0, x = var_933)[name = tensor("transpose_169")]; + tensor mul_11 = mul(x = query_23, y = var_11)[name = tensor("mul_11")]; + tensor matmul_11_transpose_y_0 = const()[name = tensor("matmul_11_transpose_y_0"), val = tensor(true)]; + tensor matmul_11_transpose_x_0 = const()[name = tensor("matmul_11_transpose_x_0"), val = tensor(false)]; + tensor matmul_11 = matmul(transpose_x = matmul_11_transpose_x_0, transpose_y = matmul_11_transpose_y_0, x = mul_11, y = key_23)[name = tensor("matmul_11")]; + tensor add_11 = add(x = matmul_11, y = attention_mask_27)[name = tensor("add_11")]; + tensor softmax_11_axis_0 = const()[name = tensor("softmax_11_axis_0"), val = tensor(-1)]; + tensor softmax_11 = softmax(axis = softmax_11_axis_0, x = add_11)[name = tensor("softmax_11")]; + tensor attn_output_45_transpose_x_0 = const()[name = tensor("attn_output_45_transpose_x_0"), val = tensor(false)]; + tensor attn_output_45_transpose_y_0 = const()[name = tensor("attn_output_45_transpose_y_0"), val = tensor(false)]; + tensor attn_output_45 = matmul(transpose_x = attn_output_45_transpose_x_0, transpose_y = attn_output_45_transpose_y_0, x = softmax_11, y = value_23)[name = tensor("attn_output_45")]; + tensor var_961_perm_0 = const()[name = tensor("op_961_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_963 = const()[name = tensor("op_963"), val = tensor([1, 2, -1])]; + tensor var_961 = transpose(perm = var_961_perm_0, x = attn_output_45)[name = tensor("transpose_168")]; + tensor var_964 = reshape(shape = var_963, x = var_961)[name = tensor("op_964")]; + tensor input_135 = linear(bias = decoder_layers_5_encoder_attn_out_proj_bias, weight = decoder_layers_5_encoder_attn_out_proj_weight_palettized, x = var_964)[name = tensor("linear_57")]; + tensor input_137 = add(x = input_131, y = input_135)[name = tensor("input_137")]; + tensor input_139_axes_0 = const()[name = tensor("input_139_axes_0"), val = tensor([-1])]; + tensor input_139 = layer_norm(axes = input_139_axes_0, beta = decoder_layers_5_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_5_final_layer_norm_weight, x = input_137)[name = tensor("input_139")]; + tensor input_141 = linear(bias = decoder_layers_5_fc1_bias, weight = decoder_layers_5_fc1_weight_palettized, x = input_139)[name = tensor("linear_58")]; + tensor input_143 = relu(x = input_141)[name = tensor("input_143")]; + tensor input_147 = linear(bias = decoder_layers_5_fc2_bias, weight = decoder_layers_5_fc2_weight_palettized, x = input_143)[name = tensor("linear_59")]; + tensor input_149 = add(x = input_137, y = input_147)[name = tensor("input_149")]; + tensor hidden_states_61_axes_0 = const()[name = tensor("hidden_states_61_axes_0"), val = tensor([-1])]; + tensor hidden_states_61 = layer_norm(axes = hidden_states_61_axes_0, beta = decoder_layers_6_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_6_self_attn_layer_norm_weight, x = input_149)[name = tensor("hidden_states_61")]; + tensor var_1014 = linear(bias = decoder_layers_6_self_attn_q_proj_bias, weight = decoder_layers_6_self_attn_q_proj_weight_palettized, x = hidden_states_61)[name = tensor("linear_60")]; + tensor var_1015 = const()[name = tensor("op_1015"), val = tensor([1, 2, -1, 64])]; + tensor var_1016 = reshape(shape = var_1015, x = var_1014)[name = tensor("op_1016")]; + tensor key_states_49 = linear(bias = decoder_layers_6_self_attn_k_proj_bias, weight = decoder_layers_6_self_attn_k_proj_weight_palettized, x = hidden_states_61)[name = tensor("linear_61")]; + tensor value_states_49 = linear(bias = decoder_layers_6_self_attn_v_proj_bias, weight = decoder_layers_6_self_attn_v_proj_weight_palettized, x = hidden_states_61)[name = tensor("linear_62")]; + tensor var_1024 = const()[name = tensor("op_1024"), val = tensor([1, 2, -1, 64])]; + tensor var_1025 = reshape(shape = var_1024, x = key_states_49)[name = tensor("op_1025")]; + tensor var_1027 = const()[name = tensor("op_1027"), val = tensor([1, 2, -1, 64])]; + tensor var_1028 = reshape(shape = var_1027, x = value_states_49)[name = tensor("op_1028")]; + tensor value_states_51_perm_0 = const()[name = tensor("value_states_51_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_25_interleave_0 = const()[name = tensor("key_25_interleave_0"), val = tensor(false)]; + tensor const_129 = const()[name = tensor("const_129"), val = tensor(1)]; + tensor key_25 = concat(axis = const_129, interleave = key_25_interleave_0, values = var_1025)[name = tensor("key_25")]; + tensor value_25_interleave_0 = const()[name = tensor("value_25_interleave_0"), val = tensor(false)]; + tensor value_states_51 = transpose(perm = value_states_51_perm_0, x = var_1028)[name = tensor("transpose_167")]; + tensor value_25 = concat(axis = var_13, interleave = value_25_interleave_0, values = value_states_51)[name = tensor("value_25")]; + tensor mul_12 = mul(x = var_1016, y = var_11)[name = tensor("mul_12")]; + tensor matmul_12_transpose_y_0 = const()[name = tensor("matmul_12_transpose_y_0"), val = tensor(true)]; + tensor matmul_12_transpose_x_0 = const()[name = tensor("matmul_12_transpose_x_0"), val = tensor(false)]; + tensor transpose_84_perm_0 = const()[name = tensor("transpose_84_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_85_perm_0 = const()[name = tensor("transpose_85_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_85 = transpose(perm = transpose_85_perm_0, x = key_25)[name = tensor("transpose_165")]; + tensor transpose_84 = transpose(perm = transpose_84_perm_0, x = mul_12)[name = tensor("transpose_166")]; + tensor matmul_12 = matmul(transpose_x = matmul_12_transpose_x_0, transpose_y = matmul_12_transpose_y_0, x = transpose_84, y = transpose_85)[name = tensor("matmul_12")]; + tensor add_12 = add(x = matmul_12, y = reshape_4)[name = tensor("add_12")]; + tensor softmax_12_axis_0 = const()[name = tensor("softmax_12_axis_0"), val = tensor(-1)]; + tensor softmax_12 = softmax(axis = softmax_12_axis_0, x = add_12)[name = tensor("softmax_12")]; + tensor attn_output_49_transpose_x_0 = const()[name = tensor("attn_output_49_transpose_x_0"), val = tensor(false)]; + tensor attn_output_49_transpose_y_0 = const()[name = tensor("attn_output_49_transpose_y_0"), val = tensor(false)]; + tensor attn_output_49 = matmul(transpose_x = attn_output_49_transpose_x_0, transpose_y = attn_output_49_transpose_y_0, x = softmax_12, y = value_25)[name = tensor("attn_output_49")]; + tensor var_1044_perm_0 = const()[name = tensor("op_1044_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1046 = const()[name = tensor("op_1046"), val = tensor([1, 2, -1])]; + tensor var_1044 = transpose(perm = var_1044_perm_0, x = attn_output_49)[name = tensor("transpose_164")]; + tensor var_1047 = reshape(shape = var_1046, x = var_1044)[name = tensor("op_1047")]; + tensor input_153 = linear(bias = decoder_layers_6_self_attn_out_proj_bias, weight = decoder_layers_6_self_attn_out_proj_weight_palettized, x = var_1047)[name = tensor("linear_63")]; + tensor input_155 = add(x = input_149, y = input_153)[name = tensor("input_155")]; + tensor hidden_states_65_axes_0 = const()[name = tensor("hidden_states_65_axes_0"), val = tensor([-1])]; + tensor hidden_states_65 = layer_norm(axes = hidden_states_65_axes_0, beta = decoder_layers_6_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_6_encoder_attn_layer_norm_weight, x = input_155)[name = tensor("hidden_states_65")]; + tensor var_1071 = linear(bias = decoder_layers_6_encoder_attn_q_proj_bias, weight = decoder_layers_6_encoder_attn_q_proj_weight_palettized, x = hidden_states_65)[name = tensor("linear_64")]; + tensor var_1072 = const()[name = tensor("op_1072"), val = tensor([1, 2, -1, 64])]; + tensor var_1073 = reshape(shape = var_1072, x = var_1071)[name = tensor("op_1073")]; + tensor query_27_perm_0 = const()[name = tensor("query_27_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_53 = linear(bias = decoder_layers_6_encoder_attn_k_proj_bias, weight = decoder_layers_6_encoder_attn_k_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_65")]; + tensor value_states_53 = linear(bias = decoder_layers_6_encoder_attn_v_proj_bias, weight = decoder_layers_6_encoder_attn_v_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_66")]; + tensor concat_22x = const()[name = tensor("concat_22x"), val = tensor([1, -1, 16, 64])]; + tensor var_1082 = reshape(shape = concat_22x, x = key_states_53)[name = tensor("op_1082")]; + tensor key_states_55_perm_0 = const()[name = tensor("key_states_55_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_23x = const()[name = tensor("concat_23x"), val = tensor([1, -1, 16, 64])]; + tensor var_1085 = reshape(shape = concat_23x, x = value_states_53)[name = tensor("op_1085")]; + tensor value_states_55_perm_0 = const()[name = tensor("value_states_55_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_27_interleave_0 = const()[name = tensor("key_27_interleave_0"), val = tensor(false)]; + tensor key_states_55 = transpose(perm = key_states_55_perm_0, x = var_1082)[name = tensor("transpose_163")]; + tensor key_27 = concat(axis = var_13, interleave = key_27_interleave_0, values = key_states_55)[name = tensor("key_27")]; + tensor value_27_interleave_0 = const()[name = tensor("value_27_interleave_0"), val = tensor(false)]; + tensor value_states_55 = transpose(perm = value_states_55_perm_0, x = var_1085)[name = tensor("transpose_162")]; + tensor value_27 = concat(axis = var_13, interleave = value_27_interleave_0, values = value_states_55)[name = tensor("value_27")]; + tensor var_1095_shape = shape(x = key_27)[name = tensor("op_1095_shape")]; + tensor gather_15_indices_0 = const()[name = tensor("gather_15_indices_0"), val = tensor(2)]; + tensor gather_15_axis_0 = const()[name = tensor("gather_15_axis_0"), val = tensor(0)]; + tensor gather_15_batch_dims_0 = const()[name = tensor("gather_15_batch_dims_0"), val = tensor(0)]; + tensor gather_15 = gather(axis = gather_15_axis_0, batch_dims = gather_15_batch_dims_0, indices = gather_15_indices_0, x = var_1095_shape)[name = tensor("gather_15")]; + tensor concat_24_values0_0 = const()[name = tensor("concat_24_values0_0"), val = tensor(0)]; + tensor concat_24_values1_0 = const()[name = tensor("concat_24_values1_0"), val = tensor(0)]; + tensor concat_24_values2_0 = const()[name = tensor("concat_24_values2_0"), val = tensor(0)]; + tensor concat_24_axis_0 = const()[name = tensor("concat_24_axis_0"), val = tensor(0)]; + tensor concat_24_interleave_0 = const()[name = tensor("concat_24_interleave_0"), val = tensor(false)]; + tensor concat_24 = concat(axis = concat_24_axis_0, interleave = concat_24_interleave_0, values = (concat_24_values0_0, concat_24_values1_0, concat_24_values2_0, gather_15))[name = tensor("concat_24")]; + tensor attention_mask_31_begin_0 = const()[name = tensor("attention_mask_31_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_31_end_mask_0 = const()[name = tensor("attention_mask_31_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_31 = slice_by_index(begin = attention_mask_31_begin_0, end = concat_24, end_mask = attention_mask_31_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_31")]; + tensor query_27 = transpose(perm = query_27_perm_0, x = var_1073)[name = tensor("transpose_161")]; + tensor mul_13 = mul(x = query_27, y = var_11)[name = tensor("mul_13")]; + tensor matmul_13_transpose_y_0 = const()[name = tensor("matmul_13_transpose_y_0"), val = tensor(true)]; + tensor matmul_13_transpose_x_0 = const()[name = tensor("matmul_13_transpose_x_0"), val = tensor(false)]; + tensor matmul_13 = matmul(transpose_x = matmul_13_transpose_x_0, transpose_y = matmul_13_transpose_y_0, x = mul_13, y = key_27)[name = tensor("matmul_13")]; + tensor add_13 = add(x = matmul_13, y = attention_mask_31)[name = tensor("add_13")]; + tensor softmax_13_axis_0 = const()[name = tensor("softmax_13_axis_0"), val = tensor(-1)]; + tensor softmax_13 = softmax(axis = softmax_13_axis_0, x = add_13)[name = tensor("softmax_13")]; + tensor attn_output_53_transpose_x_0 = const()[name = tensor("attn_output_53_transpose_x_0"), val = tensor(false)]; + tensor attn_output_53_transpose_y_0 = const()[name = tensor("attn_output_53_transpose_y_0"), val = tensor(false)]; + tensor attn_output_53 = matmul(transpose_x = attn_output_53_transpose_x_0, transpose_y = attn_output_53_transpose_y_0, x = softmax_13, y = value_27)[name = tensor("attn_output_53")]; + tensor var_1101_perm_0 = const()[name = tensor("op_1101_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1103 = const()[name = tensor("op_1103"), val = tensor([1, 2, -1])]; + tensor var_1101 = transpose(perm = var_1101_perm_0, x = attn_output_53)[name = tensor("transpose_160")]; + tensor var_1104 = reshape(shape = var_1103, x = var_1101)[name = tensor("op_1104")]; + tensor input_159 = linear(bias = decoder_layers_6_encoder_attn_out_proj_bias, weight = decoder_layers_6_encoder_attn_out_proj_weight_palettized, x = var_1104)[name = tensor("linear_67")]; + tensor input_161 = add(x = input_155, y = input_159)[name = tensor("input_161")]; + tensor input_163_axes_0 = const()[name = tensor("input_163_axes_0"), val = tensor([-1])]; + tensor input_163 = layer_norm(axes = input_163_axes_0, beta = decoder_layers_6_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_6_final_layer_norm_weight, x = input_161)[name = tensor("input_163")]; + tensor input_165 = linear(bias = decoder_layers_6_fc1_bias, weight = decoder_layers_6_fc1_weight_palettized, x = input_163)[name = tensor("linear_68")]; + tensor input_167 = relu(x = input_165)[name = tensor("input_167")]; + tensor input_171 = linear(bias = decoder_layers_6_fc2_bias, weight = decoder_layers_6_fc2_weight_palettized, x = input_167)[name = tensor("linear_69")]; + tensor input_173 = add(x = input_161, y = input_171)[name = tensor("input_173")]; + tensor hidden_states_71_axes_0 = const()[name = tensor("hidden_states_71_axes_0"), val = tensor([-1])]; + tensor hidden_states_71 = layer_norm(axes = hidden_states_71_axes_0, beta = decoder_layers_7_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_7_self_attn_layer_norm_weight, x = input_173)[name = tensor("hidden_states_71")]; + tensor var_1154 = linear(bias = decoder_layers_7_self_attn_q_proj_bias, weight = decoder_layers_7_self_attn_q_proj_weight_palettized, x = hidden_states_71)[name = tensor("linear_70")]; + tensor var_1155 = const()[name = tensor("op_1155"), val = tensor([1, 2, -1, 64])]; + tensor var_1156 = reshape(shape = var_1155, x = var_1154)[name = tensor("op_1156")]; + tensor key_states_57 = linear(bias = decoder_layers_7_self_attn_k_proj_bias, weight = decoder_layers_7_self_attn_k_proj_weight_palettized, x = hidden_states_71)[name = tensor("linear_71")]; + tensor value_states_57 = linear(bias = decoder_layers_7_self_attn_v_proj_bias, weight = decoder_layers_7_self_attn_v_proj_weight_palettized, x = hidden_states_71)[name = tensor("linear_72")]; + tensor var_1164 = const()[name = tensor("op_1164"), val = tensor([1, 2, -1, 64])]; + tensor var_1165 = reshape(shape = var_1164, x = key_states_57)[name = tensor("op_1165")]; + tensor var_1167 = const()[name = tensor("op_1167"), val = tensor([1, 2, -1, 64])]; + tensor var_1168 = reshape(shape = var_1167, x = value_states_57)[name = tensor("op_1168")]; + tensor value_states_59_perm_0 = const()[name = tensor("value_states_59_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_29_interleave_0 = const()[name = tensor("key_29_interleave_0"), val = tensor(false)]; + tensor const_130 = const()[name = tensor("const_130"), val = tensor(1)]; + tensor key_29 = concat(axis = const_130, interleave = key_29_interleave_0, values = var_1165)[name = tensor("key_29")]; + tensor value_29_interleave_0 = const()[name = tensor("value_29_interleave_0"), val = tensor(false)]; + tensor value_states_59 = transpose(perm = value_states_59_perm_0, x = var_1168)[name = tensor("transpose_159")]; + tensor value_29 = concat(axis = var_13, interleave = value_29_interleave_0, values = value_states_59)[name = tensor("value_29")]; + tensor mul_14 = mul(x = var_1156, y = var_11)[name = tensor("mul_14")]; + tensor matmul_14_transpose_y_0 = const()[name = tensor("matmul_14_transpose_y_0"), val = tensor(true)]; + tensor matmul_14_transpose_x_0 = const()[name = tensor("matmul_14_transpose_x_0"), val = tensor(false)]; + tensor transpose_86_perm_0 = const()[name = tensor("transpose_86_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_87_perm_0 = const()[name = tensor("transpose_87_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_87 = transpose(perm = transpose_87_perm_0, x = key_29)[name = tensor("transpose_157")]; + tensor transpose_86 = transpose(perm = transpose_86_perm_0, x = mul_14)[name = tensor("transpose_158")]; + tensor matmul_14 = matmul(transpose_x = matmul_14_transpose_x_0, transpose_y = matmul_14_transpose_y_0, x = transpose_86, y = transpose_87)[name = tensor("matmul_14")]; + tensor add_14 = add(x = matmul_14, y = reshape_4)[name = tensor("add_14")]; + tensor softmax_14_axis_0 = const()[name = tensor("softmax_14_axis_0"), val = tensor(-1)]; + tensor softmax_14 = softmax(axis = softmax_14_axis_0, x = add_14)[name = tensor("softmax_14")]; + tensor attn_output_57_transpose_x_0 = const()[name = tensor("attn_output_57_transpose_x_0"), val = tensor(false)]; + tensor attn_output_57_transpose_y_0 = const()[name = tensor("attn_output_57_transpose_y_0"), val = tensor(false)]; + tensor attn_output_57 = matmul(transpose_x = attn_output_57_transpose_x_0, transpose_y = attn_output_57_transpose_y_0, x = softmax_14, y = value_29)[name = tensor("attn_output_57")]; + tensor var_1184_perm_0 = const()[name = tensor("op_1184_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1186 = const()[name = tensor("op_1186"), val = tensor([1, 2, -1])]; + tensor var_1184 = transpose(perm = var_1184_perm_0, x = attn_output_57)[name = tensor("transpose_156")]; + tensor var_1187 = reshape(shape = var_1186, x = var_1184)[name = tensor("op_1187")]; + tensor input_177 = linear(bias = decoder_layers_7_self_attn_out_proj_bias, weight = decoder_layers_7_self_attn_out_proj_weight_palettized, x = var_1187)[name = tensor("linear_73")]; + tensor input_179 = add(x = input_173, y = input_177)[name = tensor("input_179")]; + tensor hidden_states_75_axes_0 = const()[name = tensor("hidden_states_75_axes_0"), val = tensor([-1])]; + tensor hidden_states_75 = layer_norm(axes = hidden_states_75_axes_0, beta = decoder_layers_7_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_7_encoder_attn_layer_norm_weight, x = input_179)[name = tensor("hidden_states_75")]; + tensor var_1211 = linear(bias = decoder_layers_7_encoder_attn_q_proj_bias, weight = decoder_layers_7_encoder_attn_q_proj_weight_palettized, x = hidden_states_75)[name = tensor("linear_74")]; + tensor var_1212 = const()[name = tensor("op_1212"), val = tensor([1, 2, -1, 64])]; + tensor var_1213 = reshape(shape = var_1212, x = var_1211)[name = tensor("op_1213")]; + tensor query_31_perm_0 = const()[name = tensor("query_31_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_61 = linear(bias = decoder_layers_7_encoder_attn_k_proj_bias, weight = decoder_layers_7_encoder_attn_k_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_75")]; + tensor value_states_61 = linear(bias = decoder_layers_7_encoder_attn_v_proj_bias, weight = decoder_layers_7_encoder_attn_v_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_76")]; + tensor concat_25x = const()[name = tensor("concat_25x"), val = tensor([1, -1, 16, 64])]; + tensor var_1222 = reshape(shape = concat_25x, x = key_states_61)[name = tensor("op_1222")]; + tensor key_states_63_perm_0 = const()[name = tensor("key_states_63_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_26x = const()[name = tensor("concat_26x"), val = tensor([1, -1, 16, 64])]; + tensor var_1225 = reshape(shape = concat_26x, x = value_states_61)[name = tensor("op_1225")]; + tensor value_states_63_perm_0 = const()[name = tensor("value_states_63_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_31_interleave_0 = const()[name = tensor("key_31_interleave_0"), val = tensor(false)]; + tensor key_states_63 = transpose(perm = key_states_63_perm_0, x = var_1222)[name = tensor("transpose_155")]; + tensor key_31 = concat(axis = var_13, interleave = key_31_interleave_0, values = key_states_63)[name = tensor("key_31")]; + tensor value_31_interleave_0 = const()[name = tensor("value_31_interleave_0"), val = tensor(false)]; + tensor value_states_63 = transpose(perm = value_states_63_perm_0, x = var_1225)[name = tensor("transpose_154")]; + tensor value_31 = concat(axis = var_13, interleave = value_31_interleave_0, values = value_states_63)[name = tensor("value_31")]; + tensor var_1235_shape = shape(x = key_31)[name = tensor("op_1235_shape")]; + tensor gather_17_indices_0 = const()[name = tensor("gather_17_indices_0"), val = tensor(2)]; + tensor gather_17_axis_0 = const()[name = tensor("gather_17_axis_0"), val = tensor(0)]; + tensor gather_17_batch_dims_0 = const()[name = tensor("gather_17_batch_dims_0"), val = tensor(0)]; + tensor gather_17 = gather(axis = gather_17_axis_0, batch_dims = gather_17_batch_dims_0, indices = gather_17_indices_0, x = var_1235_shape)[name = tensor("gather_17")]; + tensor concat_27_values0_0 = const()[name = tensor("concat_27_values0_0"), val = tensor(0)]; + tensor concat_27_values1_0 = const()[name = tensor("concat_27_values1_0"), val = tensor(0)]; + tensor concat_27_values2_0 = const()[name = tensor("concat_27_values2_0"), val = tensor(0)]; + tensor concat_27_axis_0 = const()[name = tensor("concat_27_axis_0"), val = tensor(0)]; + tensor concat_27_interleave_0 = const()[name = tensor("concat_27_interleave_0"), val = tensor(false)]; + tensor concat_27 = concat(axis = concat_27_axis_0, interleave = concat_27_interleave_0, values = (concat_27_values0_0, concat_27_values1_0, concat_27_values2_0, gather_17))[name = tensor("concat_27")]; + tensor attention_mask_35_begin_0 = const()[name = tensor("attention_mask_35_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_35_end_mask_0 = const()[name = tensor("attention_mask_35_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_35 = slice_by_index(begin = attention_mask_35_begin_0, end = concat_27, end_mask = attention_mask_35_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_35")]; + tensor query_31 = transpose(perm = query_31_perm_0, x = var_1213)[name = tensor("transpose_153")]; + tensor mul_15 = mul(x = query_31, y = var_11)[name = tensor("mul_15")]; + tensor matmul_15_transpose_y_0 = const()[name = tensor("matmul_15_transpose_y_0"), val = tensor(true)]; + tensor matmul_15_transpose_x_0 = const()[name = tensor("matmul_15_transpose_x_0"), val = tensor(false)]; + tensor matmul_15 = matmul(transpose_x = matmul_15_transpose_x_0, transpose_y = matmul_15_transpose_y_0, x = mul_15, y = key_31)[name = tensor("matmul_15")]; + tensor add_15 = add(x = matmul_15, y = attention_mask_35)[name = tensor("add_15")]; + tensor softmax_15_axis_0 = const()[name = tensor("softmax_15_axis_0"), val = tensor(-1)]; + tensor softmax_15 = softmax(axis = softmax_15_axis_0, x = add_15)[name = tensor("softmax_15")]; + tensor attn_output_61_transpose_x_0 = const()[name = tensor("attn_output_61_transpose_x_0"), val = tensor(false)]; + tensor attn_output_61_transpose_y_0 = const()[name = tensor("attn_output_61_transpose_y_0"), val = tensor(false)]; + tensor attn_output_61 = matmul(transpose_x = attn_output_61_transpose_x_0, transpose_y = attn_output_61_transpose_y_0, x = softmax_15, y = value_31)[name = tensor("attn_output_61")]; + tensor var_1241_perm_0 = const()[name = tensor("op_1241_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1243 = const()[name = tensor("op_1243"), val = tensor([1, 2, -1])]; + tensor var_1241 = transpose(perm = var_1241_perm_0, x = attn_output_61)[name = tensor("transpose_152")]; + tensor var_1244 = reshape(shape = var_1243, x = var_1241)[name = tensor("op_1244")]; + tensor input_183 = linear(bias = decoder_layers_7_encoder_attn_out_proj_bias, weight = decoder_layers_7_encoder_attn_out_proj_weight_palettized, x = var_1244)[name = tensor("linear_77")]; + tensor input_185 = add(x = input_179, y = input_183)[name = tensor("input_185")]; + tensor input_187_axes_0 = const()[name = tensor("input_187_axes_0"), val = tensor([-1])]; + tensor input_187 = layer_norm(axes = input_187_axes_0, beta = decoder_layers_7_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_7_final_layer_norm_weight, x = input_185)[name = tensor("input_187")]; + tensor input_189 = linear(bias = decoder_layers_7_fc1_bias, weight = decoder_layers_7_fc1_weight_palettized, x = input_187)[name = tensor("linear_78")]; + tensor input_191 = relu(x = input_189)[name = tensor("input_191")]; + tensor input_195 = linear(bias = decoder_layers_7_fc2_bias, weight = decoder_layers_7_fc2_weight_palettized, x = input_191)[name = tensor("linear_79")]; + tensor input_197 = add(x = input_185, y = input_195)[name = tensor("input_197")]; + tensor hidden_states_81_axes_0 = const()[name = tensor("hidden_states_81_axes_0"), val = tensor([-1])]; + tensor hidden_states_81 = layer_norm(axes = hidden_states_81_axes_0, beta = decoder_layers_8_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_8_self_attn_layer_norm_weight, x = input_197)[name = tensor("hidden_states_81")]; + tensor var_1294 = linear(bias = decoder_layers_8_self_attn_q_proj_bias, weight = decoder_layers_8_self_attn_q_proj_weight_palettized, x = hidden_states_81)[name = tensor("linear_80")]; + tensor var_1295 = const()[name = tensor("op_1295"), val = tensor([1, 2, -1, 64])]; + tensor var_1296 = reshape(shape = var_1295, x = var_1294)[name = tensor("op_1296")]; + tensor key_states_65 = linear(bias = decoder_layers_8_self_attn_k_proj_bias, weight = decoder_layers_8_self_attn_k_proj_weight_palettized, x = hidden_states_81)[name = tensor("linear_81")]; + tensor value_states_65 = linear(bias = decoder_layers_8_self_attn_v_proj_bias, weight = decoder_layers_8_self_attn_v_proj_weight_palettized, x = hidden_states_81)[name = tensor("linear_82")]; + tensor var_1304 = const()[name = tensor("op_1304"), val = tensor([1, 2, -1, 64])]; + tensor var_1305 = reshape(shape = var_1304, x = key_states_65)[name = tensor("op_1305")]; + tensor var_1307 = const()[name = tensor("op_1307"), val = tensor([1, 2, -1, 64])]; + tensor var_1308 = reshape(shape = var_1307, x = value_states_65)[name = tensor("op_1308")]; + tensor value_states_67_perm_0 = const()[name = tensor("value_states_67_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_33_interleave_0 = const()[name = tensor("key_33_interleave_0"), val = tensor(false)]; + tensor const_131 = const()[name = tensor("const_131"), val = tensor(1)]; + tensor key_33 = concat(axis = const_131, interleave = key_33_interleave_0, values = var_1305)[name = tensor("key_33")]; + tensor value_33_interleave_0 = const()[name = tensor("value_33_interleave_0"), val = tensor(false)]; + tensor value_states_67 = transpose(perm = value_states_67_perm_0, x = var_1308)[name = tensor("transpose_151")]; + tensor value_33 = concat(axis = var_13, interleave = value_33_interleave_0, values = value_states_67)[name = tensor("value_33")]; + tensor mul_16 = mul(x = var_1296, y = var_11)[name = tensor("mul_16")]; + tensor matmul_16_transpose_y_0 = const()[name = tensor("matmul_16_transpose_y_0"), val = tensor(true)]; + tensor matmul_16_transpose_x_0 = const()[name = tensor("matmul_16_transpose_x_0"), val = tensor(false)]; + tensor transpose_88_perm_0 = const()[name = tensor("transpose_88_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_89_perm_0 = const()[name = tensor("transpose_89_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_89 = transpose(perm = transpose_89_perm_0, x = key_33)[name = tensor("transpose_149")]; + tensor transpose_88 = transpose(perm = transpose_88_perm_0, x = mul_16)[name = tensor("transpose_150")]; + tensor matmul_16 = matmul(transpose_x = matmul_16_transpose_x_0, transpose_y = matmul_16_transpose_y_0, x = transpose_88, y = transpose_89)[name = tensor("matmul_16")]; + tensor add_16 = add(x = matmul_16, y = reshape_4)[name = tensor("add_16")]; + tensor softmax_16_axis_0 = const()[name = tensor("softmax_16_axis_0"), val = tensor(-1)]; + tensor softmax_16 = softmax(axis = softmax_16_axis_0, x = add_16)[name = tensor("softmax_16")]; + tensor attn_output_65_transpose_x_0 = const()[name = tensor("attn_output_65_transpose_x_0"), val = tensor(false)]; + tensor attn_output_65_transpose_y_0 = const()[name = tensor("attn_output_65_transpose_y_0"), val = tensor(false)]; + tensor attn_output_65 = matmul(transpose_x = attn_output_65_transpose_x_0, transpose_y = attn_output_65_transpose_y_0, x = softmax_16, y = value_33)[name = tensor("attn_output_65")]; + tensor var_1324_perm_0 = const()[name = tensor("op_1324_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1326 = const()[name = tensor("op_1326"), val = tensor([1, 2, -1])]; + tensor var_1324 = transpose(perm = var_1324_perm_0, x = attn_output_65)[name = tensor("transpose_148")]; + tensor var_1327 = reshape(shape = var_1326, x = var_1324)[name = tensor("op_1327")]; + tensor input_201 = linear(bias = decoder_layers_8_self_attn_out_proj_bias, weight = decoder_layers_8_self_attn_out_proj_weight_palettized, x = var_1327)[name = tensor("linear_83")]; + tensor input_203 = add(x = input_197, y = input_201)[name = tensor("input_203")]; + tensor hidden_states_85_axes_0 = const()[name = tensor("hidden_states_85_axes_0"), val = tensor([-1])]; + tensor hidden_states_85 = layer_norm(axes = hidden_states_85_axes_0, beta = decoder_layers_8_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_8_encoder_attn_layer_norm_weight, x = input_203)[name = tensor("hidden_states_85")]; + tensor var_1351 = linear(bias = decoder_layers_8_encoder_attn_q_proj_bias, weight = decoder_layers_8_encoder_attn_q_proj_weight_palettized, x = hidden_states_85)[name = tensor("linear_84")]; + tensor var_1352 = const()[name = tensor("op_1352"), val = tensor([1, 2, -1, 64])]; + tensor var_1353 = reshape(shape = var_1352, x = var_1351)[name = tensor("op_1353")]; + tensor query_35_perm_0 = const()[name = tensor("query_35_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_69 = linear(bias = decoder_layers_8_encoder_attn_k_proj_bias, weight = decoder_layers_8_encoder_attn_k_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_85")]; + tensor value_states_69 = linear(bias = decoder_layers_8_encoder_attn_v_proj_bias, weight = decoder_layers_8_encoder_attn_v_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_86")]; + tensor concat_28x = const()[name = tensor("concat_28x"), val = tensor([1, -1, 16, 64])]; + tensor var_1362 = reshape(shape = concat_28x, x = key_states_69)[name = tensor("op_1362")]; + tensor key_states_71_perm_0 = const()[name = tensor("key_states_71_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_29x = const()[name = tensor("concat_29x"), val = tensor([1, -1, 16, 64])]; + tensor var_1365 = reshape(shape = concat_29x, x = value_states_69)[name = tensor("op_1365")]; + tensor value_states_71_perm_0 = const()[name = tensor("value_states_71_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_35_interleave_0 = const()[name = tensor("key_35_interleave_0"), val = tensor(false)]; + tensor key_states_71 = transpose(perm = key_states_71_perm_0, x = var_1362)[name = tensor("transpose_147")]; + tensor key_35 = concat(axis = var_13, interleave = key_35_interleave_0, values = key_states_71)[name = tensor("key_35")]; + tensor value_35_interleave_0 = const()[name = tensor("value_35_interleave_0"), val = tensor(false)]; + tensor value_states_71 = transpose(perm = value_states_71_perm_0, x = var_1365)[name = tensor("transpose_146")]; + tensor value_35 = concat(axis = var_13, interleave = value_35_interleave_0, values = value_states_71)[name = tensor("value_35")]; + tensor var_1375_shape = shape(x = key_35)[name = tensor("op_1375_shape")]; + tensor gather_19_indices_0 = const()[name = tensor("gather_19_indices_0"), val = tensor(2)]; + tensor gather_19_axis_0 = const()[name = tensor("gather_19_axis_0"), val = tensor(0)]; + tensor gather_19_batch_dims_0 = const()[name = tensor("gather_19_batch_dims_0"), val = tensor(0)]; + tensor gather_19 = gather(axis = gather_19_axis_0, batch_dims = gather_19_batch_dims_0, indices = gather_19_indices_0, x = var_1375_shape)[name = tensor("gather_19")]; + tensor concat_30_values0_0 = const()[name = tensor("concat_30_values0_0"), val = tensor(0)]; + tensor concat_30_values1_0 = const()[name = tensor("concat_30_values1_0"), val = tensor(0)]; + tensor concat_30_values2_0 = const()[name = tensor("concat_30_values2_0"), val = tensor(0)]; + tensor concat_30_axis_0 = const()[name = tensor("concat_30_axis_0"), val = tensor(0)]; + tensor concat_30_interleave_0 = const()[name = tensor("concat_30_interleave_0"), val = tensor(false)]; + tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (concat_30_values0_0, concat_30_values1_0, concat_30_values2_0, gather_19))[name = tensor("concat_30")]; + tensor attention_mask_39_begin_0 = const()[name = tensor("attention_mask_39_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_39_end_mask_0 = const()[name = tensor("attention_mask_39_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_39 = slice_by_index(begin = attention_mask_39_begin_0, end = concat_30, end_mask = attention_mask_39_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_39")]; + tensor query_35 = transpose(perm = query_35_perm_0, x = var_1353)[name = tensor("transpose_145")]; + tensor mul_17 = mul(x = query_35, y = var_11)[name = tensor("mul_17")]; + tensor matmul_17_transpose_y_0 = const()[name = tensor("matmul_17_transpose_y_0"), val = tensor(true)]; + tensor matmul_17_transpose_x_0 = const()[name = tensor("matmul_17_transpose_x_0"), val = tensor(false)]; + tensor matmul_17 = matmul(transpose_x = matmul_17_transpose_x_0, transpose_y = matmul_17_transpose_y_0, x = mul_17, y = key_35)[name = tensor("matmul_17")]; + tensor add_17 = add(x = matmul_17, y = attention_mask_39)[name = tensor("add_17")]; + tensor softmax_17_axis_0 = const()[name = tensor("softmax_17_axis_0"), val = tensor(-1)]; + tensor softmax_17 = softmax(axis = softmax_17_axis_0, x = add_17)[name = tensor("softmax_17")]; + tensor attn_output_69_transpose_x_0 = const()[name = tensor("attn_output_69_transpose_x_0"), val = tensor(false)]; + tensor attn_output_69_transpose_y_0 = const()[name = tensor("attn_output_69_transpose_y_0"), val = tensor(false)]; + tensor attn_output_69 = matmul(transpose_x = attn_output_69_transpose_x_0, transpose_y = attn_output_69_transpose_y_0, x = softmax_17, y = value_35)[name = tensor("attn_output_69")]; + tensor var_1381_perm_0 = const()[name = tensor("op_1381_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1383 = const()[name = tensor("op_1383"), val = tensor([1, 2, -1])]; + tensor var_1381 = transpose(perm = var_1381_perm_0, x = attn_output_69)[name = tensor("transpose_144")]; + tensor var_1384 = reshape(shape = var_1383, x = var_1381)[name = tensor("op_1384")]; + tensor input_207 = linear(bias = decoder_layers_8_encoder_attn_out_proj_bias, weight = decoder_layers_8_encoder_attn_out_proj_weight_palettized, x = var_1384)[name = tensor("linear_87")]; + tensor input_209 = add(x = input_203, y = input_207)[name = tensor("input_209")]; + tensor input_211_axes_0 = const()[name = tensor("input_211_axes_0"), val = tensor([-1])]; + tensor input_211 = layer_norm(axes = input_211_axes_0, beta = decoder_layers_8_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_8_final_layer_norm_weight, x = input_209)[name = tensor("input_211")]; + tensor input_213 = linear(bias = decoder_layers_8_fc1_bias, weight = decoder_layers_8_fc1_weight_palettized, x = input_211)[name = tensor("linear_88")]; + tensor input_215 = relu(x = input_213)[name = tensor("input_215")]; + tensor input_219 = linear(bias = decoder_layers_8_fc2_bias, weight = decoder_layers_8_fc2_weight_palettized, x = input_215)[name = tensor("linear_89")]; + tensor input_221 = add(x = input_209, y = input_219)[name = tensor("input_221")]; + tensor hidden_states_91_axes_0 = const()[name = tensor("hidden_states_91_axes_0"), val = tensor([-1])]; + tensor hidden_states_91 = layer_norm(axes = hidden_states_91_axes_0, beta = decoder_layers_9_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_9_self_attn_layer_norm_weight, x = input_221)[name = tensor("hidden_states_91")]; + tensor var_1434 = linear(bias = decoder_layers_9_self_attn_q_proj_bias, weight = decoder_layers_9_self_attn_q_proj_weight_palettized, x = hidden_states_91)[name = tensor("linear_90")]; + tensor var_1435 = const()[name = tensor("op_1435"), val = tensor([1, 2, -1, 64])]; + tensor var_1436 = reshape(shape = var_1435, x = var_1434)[name = tensor("op_1436")]; + tensor key_states_73 = linear(bias = decoder_layers_9_self_attn_k_proj_bias, weight = decoder_layers_9_self_attn_k_proj_weight_palettized, x = hidden_states_91)[name = tensor("linear_91")]; + tensor value_states_73 = linear(bias = decoder_layers_9_self_attn_v_proj_bias, weight = decoder_layers_9_self_attn_v_proj_weight_palettized, x = hidden_states_91)[name = tensor("linear_92")]; + tensor var_1444 = const()[name = tensor("op_1444"), val = tensor([1, 2, -1, 64])]; + tensor var_1445 = reshape(shape = var_1444, x = key_states_73)[name = tensor("op_1445")]; + tensor var_1447 = const()[name = tensor("op_1447"), val = tensor([1, 2, -1, 64])]; + tensor var_1448 = reshape(shape = var_1447, x = value_states_73)[name = tensor("op_1448")]; + tensor value_states_75_perm_0 = const()[name = tensor("value_states_75_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_37_interleave_0 = const()[name = tensor("key_37_interleave_0"), val = tensor(false)]; + tensor const_132 = const()[name = tensor("const_132"), val = tensor(1)]; + tensor key_37 = concat(axis = const_132, interleave = key_37_interleave_0, values = var_1445)[name = tensor("key_37")]; + tensor value_37_interleave_0 = const()[name = tensor("value_37_interleave_0"), val = tensor(false)]; + tensor value_states_75 = transpose(perm = value_states_75_perm_0, x = var_1448)[name = tensor("transpose_143")]; + tensor value_37 = concat(axis = var_13, interleave = value_37_interleave_0, values = value_states_75)[name = tensor("value_37")]; + tensor mul_18 = mul(x = var_1436, y = var_11)[name = tensor("mul_18")]; + tensor matmul_18_transpose_y_0 = const()[name = tensor("matmul_18_transpose_y_0"), val = tensor(true)]; + tensor matmul_18_transpose_x_0 = const()[name = tensor("matmul_18_transpose_x_0"), val = tensor(false)]; + tensor transpose_90_perm_0 = const()[name = tensor("transpose_90_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_91_perm_0 = const()[name = tensor("transpose_91_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_91 = transpose(perm = transpose_91_perm_0, x = key_37)[name = tensor("transpose_141")]; + tensor transpose_90 = transpose(perm = transpose_90_perm_0, x = mul_18)[name = tensor("transpose_142")]; + tensor matmul_18 = matmul(transpose_x = matmul_18_transpose_x_0, transpose_y = matmul_18_transpose_y_0, x = transpose_90, y = transpose_91)[name = tensor("matmul_18")]; + tensor add_18 = add(x = matmul_18, y = reshape_4)[name = tensor("add_18")]; + tensor softmax_18_axis_0 = const()[name = tensor("softmax_18_axis_0"), val = tensor(-1)]; + tensor softmax_18 = softmax(axis = softmax_18_axis_0, x = add_18)[name = tensor("softmax_18")]; + tensor attn_output_73_transpose_x_0 = const()[name = tensor("attn_output_73_transpose_x_0"), val = tensor(false)]; + tensor attn_output_73_transpose_y_0 = const()[name = tensor("attn_output_73_transpose_y_0"), val = tensor(false)]; + tensor attn_output_73 = matmul(transpose_x = attn_output_73_transpose_x_0, transpose_y = attn_output_73_transpose_y_0, x = softmax_18, y = value_37)[name = tensor("attn_output_73")]; + tensor var_1464_perm_0 = const()[name = tensor("op_1464_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1466 = const()[name = tensor("op_1466"), val = tensor([1, 2, -1])]; + tensor var_1464 = transpose(perm = var_1464_perm_0, x = attn_output_73)[name = tensor("transpose_140")]; + tensor var_1467 = reshape(shape = var_1466, x = var_1464)[name = tensor("op_1467")]; + tensor input_225 = linear(bias = decoder_layers_9_self_attn_out_proj_bias, weight = decoder_layers_9_self_attn_out_proj_weight_palettized, x = var_1467)[name = tensor("linear_93")]; + tensor input_227 = add(x = input_221, y = input_225)[name = tensor("input_227")]; + tensor hidden_states_95_axes_0 = const()[name = tensor("hidden_states_95_axes_0"), val = tensor([-1])]; + tensor hidden_states_95 = layer_norm(axes = hidden_states_95_axes_0, beta = decoder_layers_9_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_9_encoder_attn_layer_norm_weight, x = input_227)[name = tensor("hidden_states_95")]; + tensor var_1491 = linear(bias = decoder_layers_9_encoder_attn_q_proj_bias, weight = decoder_layers_9_encoder_attn_q_proj_weight_palettized, x = hidden_states_95)[name = tensor("linear_94")]; + tensor var_1492 = const()[name = tensor("op_1492"), val = tensor([1, 2, -1, 64])]; + tensor var_1493 = reshape(shape = var_1492, x = var_1491)[name = tensor("op_1493")]; + tensor query_39_perm_0 = const()[name = tensor("query_39_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_77 = linear(bias = decoder_layers_9_encoder_attn_k_proj_bias, weight = decoder_layers_9_encoder_attn_k_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_95")]; + tensor value_states_77 = linear(bias = decoder_layers_9_encoder_attn_v_proj_bias, weight = decoder_layers_9_encoder_attn_v_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_96")]; + tensor concat_31x = const()[name = tensor("concat_31x"), val = tensor([1, -1, 16, 64])]; + tensor var_1502 = reshape(shape = concat_31x, x = key_states_77)[name = tensor("op_1502")]; + tensor key_states_79_perm_0 = const()[name = tensor("key_states_79_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_32x = const()[name = tensor("concat_32x"), val = tensor([1, -1, 16, 64])]; + tensor var_1505 = reshape(shape = concat_32x, x = value_states_77)[name = tensor("op_1505")]; + tensor value_states_79_perm_0 = const()[name = tensor("value_states_79_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_39_interleave_0 = const()[name = tensor("key_39_interleave_0"), val = tensor(false)]; + tensor key_states_79 = transpose(perm = key_states_79_perm_0, x = var_1502)[name = tensor("transpose_139")]; + tensor key_39 = concat(axis = var_13, interleave = key_39_interleave_0, values = key_states_79)[name = tensor("key_39")]; + tensor value_39_interleave_0 = const()[name = tensor("value_39_interleave_0"), val = tensor(false)]; + tensor value_states_79 = transpose(perm = value_states_79_perm_0, x = var_1505)[name = tensor("transpose_138")]; + tensor value_39 = concat(axis = var_13, interleave = value_39_interleave_0, values = value_states_79)[name = tensor("value_39")]; + tensor var_1515_shape = shape(x = key_39)[name = tensor("op_1515_shape")]; + tensor gather_21_indices_0 = const()[name = tensor("gather_21_indices_0"), val = tensor(2)]; + tensor gather_21_axis_0 = const()[name = tensor("gather_21_axis_0"), val = tensor(0)]; + tensor gather_21_batch_dims_0 = const()[name = tensor("gather_21_batch_dims_0"), val = tensor(0)]; + tensor gather_21 = gather(axis = gather_21_axis_0, batch_dims = gather_21_batch_dims_0, indices = gather_21_indices_0, x = var_1515_shape)[name = tensor("gather_21")]; + tensor concat_33_values0_0 = const()[name = tensor("concat_33_values0_0"), val = tensor(0)]; + tensor concat_33_values1_0 = const()[name = tensor("concat_33_values1_0"), val = tensor(0)]; + tensor concat_33_values2_0 = const()[name = tensor("concat_33_values2_0"), val = tensor(0)]; + tensor concat_33_axis_0 = const()[name = tensor("concat_33_axis_0"), val = tensor(0)]; + tensor concat_33_interleave_0 = const()[name = tensor("concat_33_interleave_0"), val = tensor(false)]; + tensor concat_33 = concat(axis = concat_33_axis_0, interleave = concat_33_interleave_0, values = (concat_33_values0_0, concat_33_values1_0, concat_33_values2_0, gather_21))[name = tensor("concat_33")]; + tensor attention_mask_43_begin_0 = const()[name = tensor("attention_mask_43_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_43_end_mask_0 = const()[name = tensor("attention_mask_43_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_43 = slice_by_index(begin = attention_mask_43_begin_0, end = concat_33, end_mask = attention_mask_43_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_43")]; + tensor query_39 = transpose(perm = query_39_perm_0, x = var_1493)[name = tensor("transpose_137")]; + tensor mul_19 = mul(x = query_39, y = var_11)[name = tensor("mul_19")]; + tensor matmul_19_transpose_y_0 = const()[name = tensor("matmul_19_transpose_y_0"), val = tensor(true)]; + tensor matmul_19_transpose_x_0 = const()[name = tensor("matmul_19_transpose_x_0"), val = tensor(false)]; + tensor matmul_19 = matmul(transpose_x = matmul_19_transpose_x_0, transpose_y = matmul_19_transpose_y_0, x = mul_19, y = key_39)[name = tensor("matmul_19")]; + tensor add_19 = add(x = matmul_19, y = attention_mask_43)[name = tensor("add_19")]; + tensor softmax_19_axis_0 = const()[name = tensor("softmax_19_axis_0"), val = tensor(-1)]; + tensor softmax_19 = softmax(axis = softmax_19_axis_0, x = add_19)[name = tensor("softmax_19")]; + tensor attn_output_77_transpose_x_0 = const()[name = tensor("attn_output_77_transpose_x_0"), val = tensor(false)]; + tensor attn_output_77_transpose_y_0 = const()[name = tensor("attn_output_77_transpose_y_0"), val = tensor(false)]; + tensor attn_output_77 = matmul(transpose_x = attn_output_77_transpose_x_0, transpose_y = attn_output_77_transpose_y_0, x = softmax_19, y = value_39)[name = tensor("attn_output_77")]; + tensor var_1521_perm_0 = const()[name = tensor("op_1521_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1523 = const()[name = tensor("op_1523"), val = tensor([1, 2, -1])]; + tensor var_1521 = transpose(perm = var_1521_perm_0, x = attn_output_77)[name = tensor("transpose_136")]; + tensor var_1524 = reshape(shape = var_1523, x = var_1521)[name = tensor("op_1524")]; + tensor input_231 = linear(bias = decoder_layers_9_encoder_attn_out_proj_bias, weight = decoder_layers_9_encoder_attn_out_proj_weight_palettized, x = var_1524)[name = tensor("linear_97")]; + tensor input_233 = add(x = input_227, y = input_231)[name = tensor("input_233")]; + tensor input_235_axes_0 = const()[name = tensor("input_235_axes_0"), val = tensor([-1])]; + tensor input_235 = layer_norm(axes = input_235_axes_0, beta = decoder_layers_9_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_9_final_layer_norm_weight, x = input_233)[name = tensor("input_235")]; + tensor input_237 = linear(bias = decoder_layers_9_fc1_bias, weight = decoder_layers_9_fc1_weight_palettized, x = input_235)[name = tensor("linear_98")]; + tensor input_239 = relu(x = input_237)[name = tensor("input_239")]; + tensor input_243 = linear(bias = decoder_layers_9_fc2_bias, weight = decoder_layers_9_fc2_weight_palettized, x = input_239)[name = tensor("linear_99")]; + tensor input_245 = add(x = input_233, y = input_243)[name = tensor("input_245")]; + tensor hidden_states_101_axes_0 = const()[name = tensor("hidden_states_101_axes_0"), val = tensor([-1])]; + tensor hidden_states_101 = layer_norm(axes = hidden_states_101_axes_0, beta = decoder_layers_10_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_10_self_attn_layer_norm_weight, x = input_245)[name = tensor("hidden_states_101")]; + tensor var_1574 = linear(bias = decoder_layers_10_self_attn_q_proj_bias, weight = decoder_layers_10_self_attn_q_proj_weight_palettized, x = hidden_states_101)[name = tensor("linear_100")]; + tensor var_1575 = const()[name = tensor("op_1575"), val = tensor([1, 2, -1, 64])]; + tensor var_1576 = reshape(shape = var_1575, x = var_1574)[name = tensor("op_1576")]; + tensor key_states_81 = linear(bias = decoder_layers_10_self_attn_k_proj_bias, weight = decoder_layers_10_self_attn_k_proj_weight_palettized, x = hidden_states_101)[name = tensor("linear_101")]; + tensor value_states_81 = linear(bias = decoder_layers_10_self_attn_v_proj_bias, weight = decoder_layers_10_self_attn_v_proj_weight_palettized, x = hidden_states_101)[name = tensor("linear_102")]; + tensor var_1584 = const()[name = tensor("op_1584"), val = tensor([1, 2, -1, 64])]; + tensor var_1585 = reshape(shape = var_1584, x = key_states_81)[name = tensor("op_1585")]; + tensor var_1587 = const()[name = tensor("op_1587"), val = tensor([1, 2, -1, 64])]; + tensor var_1588 = reshape(shape = var_1587, x = value_states_81)[name = tensor("op_1588")]; + tensor value_states_83_perm_0 = const()[name = tensor("value_states_83_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_41_interleave_0 = const()[name = tensor("key_41_interleave_0"), val = tensor(false)]; + tensor const_133 = const()[name = tensor("const_133"), val = tensor(1)]; + tensor key_41 = concat(axis = const_133, interleave = key_41_interleave_0, values = var_1585)[name = tensor("key_41")]; + tensor value_41_interleave_0 = const()[name = tensor("value_41_interleave_0"), val = tensor(false)]; + tensor value_states_83 = transpose(perm = value_states_83_perm_0, x = var_1588)[name = tensor("transpose_135")]; + tensor value_41 = concat(axis = var_13, interleave = value_41_interleave_0, values = value_states_83)[name = tensor("value_41")]; + tensor mul_20 = mul(x = var_1576, y = var_11)[name = tensor("mul_20")]; + tensor matmul_20_transpose_y_0 = const()[name = tensor("matmul_20_transpose_y_0"), val = tensor(true)]; + tensor matmul_20_transpose_x_0 = const()[name = tensor("matmul_20_transpose_x_0"), val = tensor(false)]; + tensor transpose_92_perm_0 = const()[name = tensor("transpose_92_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_93_perm_0 = const()[name = tensor("transpose_93_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_93 = transpose(perm = transpose_93_perm_0, x = key_41)[name = tensor("transpose_133")]; + tensor transpose_92 = transpose(perm = transpose_92_perm_0, x = mul_20)[name = tensor("transpose_134")]; + tensor matmul_20 = matmul(transpose_x = matmul_20_transpose_x_0, transpose_y = matmul_20_transpose_y_0, x = transpose_92, y = transpose_93)[name = tensor("matmul_20")]; + tensor add_20 = add(x = matmul_20, y = reshape_4)[name = tensor("add_20")]; + tensor softmax_20_axis_0 = const()[name = tensor("softmax_20_axis_0"), val = tensor(-1)]; + tensor softmax_20 = softmax(axis = softmax_20_axis_0, x = add_20)[name = tensor("softmax_20")]; + tensor attn_output_81_transpose_x_0 = const()[name = tensor("attn_output_81_transpose_x_0"), val = tensor(false)]; + tensor attn_output_81_transpose_y_0 = const()[name = tensor("attn_output_81_transpose_y_0"), val = tensor(false)]; + tensor attn_output_81 = matmul(transpose_x = attn_output_81_transpose_x_0, transpose_y = attn_output_81_transpose_y_0, x = softmax_20, y = value_41)[name = tensor("attn_output_81")]; + tensor var_1604_perm_0 = const()[name = tensor("op_1604_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1606 = const()[name = tensor("op_1606"), val = tensor([1, 2, -1])]; + tensor var_1604 = transpose(perm = var_1604_perm_0, x = attn_output_81)[name = tensor("transpose_132")]; + tensor var_1607 = reshape(shape = var_1606, x = var_1604)[name = tensor("op_1607")]; + tensor input_249 = linear(bias = decoder_layers_10_self_attn_out_proj_bias, weight = decoder_layers_10_self_attn_out_proj_weight_palettized, x = var_1607)[name = tensor("linear_103")]; + tensor input_251 = add(x = input_245, y = input_249)[name = tensor("input_251")]; + tensor hidden_states_105_axes_0 = const()[name = tensor("hidden_states_105_axes_0"), val = tensor([-1])]; + tensor hidden_states_105 = layer_norm(axes = hidden_states_105_axes_0, beta = decoder_layers_10_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_10_encoder_attn_layer_norm_weight, x = input_251)[name = tensor("hidden_states_105")]; + tensor var_1631 = linear(bias = decoder_layers_10_encoder_attn_q_proj_bias, weight = decoder_layers_10_encoder_attn_q_proj_weight_palettized, x = hidden_states_105)[name = tensor("linear_104")]; + tensor var_1632 = const()[name = tensor("op_1632"), val = tensor([1, 2, -1, 64])]; + tensor var_1633 = reshape(shape = var_1632, x = var_1631)[name = tensor("op_1633")]; + tensor query_43_perm_0 = const()[name = tensor("query_43_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_85 = linear(bias = decoder_layers_10_encoder_attn_k_proj_bias, weight = decoder_layers_10_encoder_attn_k_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_105")]; + tensor value_states_85 = linear(bias = decoder_layers_10_encoder_attn_v_proj_bias, weight = decoder_layers_10_encoder_attn_v_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_106")]; + tensor concat_34x = const()[name = tensor("concat_34x"), val = tensor([1, -1, 16, 64])]; + tensor var_1642 = reshape(shape = concat_34x, x = key_states_85)[name = tensor("op_1642")]; + tensor key_states_87_perm_0 = const()[name = tensor("key_states_87_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_35x = const()[name = tensor("concat_35x"), val = tensor([1, -1, 16, 64])]; + tensor var_1645 = reshape(shape = concat_35x, x = value_states_85)[name = tensor("op_1645")]; + tensor value_states_87_perm_0 = const()[name = tensor("value_states_87_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_43_interleave_0 = const()[name = tensor("key_43_interleave_0"), val = tensor(false)]; + tensor key_states_87 = transpose(perm = key_states_87_perm_0, x = var_1642)[name = tensor("transpose_131")]; + tensor key_43 = concat(axis = var_13, interleave = key_43_interleave_0, values = key_states_87)[name = tensor("key_43")]; + tensor value_43_interleave_0 = const()[name = tensor("value_43_interleave_0"), val = tensor(false)]; + tensor value_states_87 = transpose(perm = value_states_87_perm_0, x = var_1645)[name = tensor("transpose_130")]; + tensor value_43 = concat(axis = var_13, interleave = value_43_interleave_0, values = value_states_87)[name = tensor("value_43")]; + tensor var_1655_shape = shape(x = key_43)[name = tensor("op_1655_shape")]; + tensor gather_23_indices_0 = const()[name = tensor("gather_23_indices_0"), val = tensor(2)]; + tensor gather_23_axis_0 = const()[name = tensor("gather_23_axis_0"), val = tensor(0)]; + tensor gather_23_batch_dims_0 = const()[name = tensor("gather_23_batch_dims_0"), val = tensor(0)]; + tensor gather_23 = gather(axis = gather_23_axis_0, batch_dims = gather_23_batch_dims_0, indices = gather_23_indices_0, x = var_1655_shape)[name = tensor("gather_23")]; + tensor concat_36_values0_0 = const()[name = tensor("concat_36_values0_0"), val = tensor(0)]; + tensor concat_36_values1_0 = const()[name = tensor("concat_36_values1_0"), val = tensor(0)]; + tensor concat_36_values2_0 = const()[name = tensor("concat_36_values2_0"), val = tensor(0)]; + tensor concat_36_axis_0 = const()[name = tensor("concat_36_axis_0"), val = tensor(0)]; + tensor concat_36_interleave_0 = const()[name = tensor("concat_36_interleave_0"), val = tensor(false)]; + tensor concat_36 = concat(axis = concat_36_axis_0, interleave = concat_36_interleave_0, values = (concat_36_values0_0, concat_36_values1_0, concat_36_values2_0, gather_23))[name = tensor("concat_36")]; + tensor attention_mask_47_begin_0 = const()[name = tensor("attention_mask_47_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_47_end_mask_0 = const()[name = tensor("attention_mask_47_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask_47 = slice_by_index(begin = attention_mask_47_begin_0, end = concat_36, end_mask = attention_mask_47_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask_47")]; + tensor query_43 = transpose(perm = query_43_perm_0, x = var_1633)[name = tensor("transpose_129")]; + tensor mul_21 = mul(x = query_43, y = var_11)[name = tensor("mul_21")]; + tensor matmul_21_transpose_y_0 = const()[name = tensor("matmul_21_transpose_y_0"), val = tensor(true)]; + tensor matmul_21_transpose_x_0 = const()[name = tensor("matmul_21_transpose_x_0"), val = tensor(false)]; + tensor matmul_21 = matmul(transpose_x = matmul_21_transpose_x_0, transpose_y = matmul_21_transpose_y_0, x = mul_21, y = key_43)[name = tensor("matmul_21")]; + tensor add_21 = add(x = matmul_21, y = attention_mask_47)[name = tensor("add_21")]; + tensor softmax_21_axis_0 = const()[name = tensor("softmax_21_axis_0"), val = tensor(-1)]; + tensor softmax_21 = softmax(axis = softmax_21_axis_0, x = add_21)[name = tensor("softmax_21")]; + tensor attn_output_85_transpose_x_0 = const()[name = tensor("attn_output_85_transpose_x_0"), val = tensor(false)]; + tensor attn_output_85_transpose_y_0 = const()[name = tensor("attn_output_85_transpose_y_0"), val = tensor(false)]; + tensor attn_output_85 = matmul(transpose_x = attn_output_85_transpose_x_0, transpose_y = attn_output_85_transpose_y_0, x = softmax_21, y = value_43)[name = tensor("attn_output_85")]; + tensor var_1661_perm_0 = const()[name = tensor("op_1661_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1663 = const()[name = tensor("op_1663"), val = tensor([1, 2, -1])]; + tensor var_1661 = transpose(perm = var_1661_perm_0, x = attn_output_85)[name = tensor("transpose_128")]; + tensor var_1664 = reshape(shape = var_1663, x = var_1661)[name = tensor("op_1664")]; + tensor input_255 = linear(bias = decoder_layers_10_encoder_attn_out_proj_bias, weight = decoder_layers_10_encoder_attn_out_proj_weight_palettized, x = var_1664)[name = tensor("linear_107")]; + tensor input_257 = add(x = input_251, y = input_255)[name = tensor("input_257")]; + tensor input_259_axes_0 = const()[name = tensor("input_259_axes_0"), val = tensor([-1])]; + tensor input_259 = layer_norm(axes = input_259_axes_0, beta = decoder_layers_10_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_10_final_layer_norm_weight, x = input_257)[name = tensor("input_259")]; + tensor input_261 = linear(bias = decoder_layers_10_fc1_bias, weight = decoder_layers_10_fc1_weight_palettized, x = input_259)[name = tensor("linear_108")]; + tensor input_263 = relu(x = input_261)[name = tensor("input_263")]; + tensor input_267 = linear(bias = decoder_layers_10_fc2_bias, weight = decoder_layers_10_fc2_weight_palettized, x = input_263)[name = tensor("linear_109")]; + tensor input_269 = add(x = input_257, y = input_267)[name = tensor("input_269")]; + tensor hidden_states_111_axes_0 = const()[name = tensor("hidden_states_111_axes_0"), val = tensor([-1])]; + tensor hidden_states_111 = layer_norm(axes = hidden_states_111_axes_0, beta = decoder_layers_11_self_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_11_self_attn_layer_norm_weight, x = input_269)[name = tensor("hidden_states_111")]; + tensor var_1714 = linear(bias = decoder_layers_11_self_attn_q_proj_bias, weight = decoder_layers_11_self_attn_q_proj_weight_palettized, x = hidden_states_111)[name = tensor("linear_110")]; + tensor var_1715 = const()[name = tensor("op_1715"), val = tensor([1, 2, -1, 64])]; + tensor var_1716 = reshape(shape = var_1715, x = var_1714)[name = tensor("op_1716")]; + tensor key_states_89 = linear(bias = decoder_layers_11_self_attn_k_proj_bias, weight = decoder_layers_11_self_attn_k_proj_weight_palettized, x = hidden_states_111)[name = tensor("linear_111")]; + tensor value_states_89 = linear(bias = decoder_layers_11_self_attn_v_proj_bias, weight = decoder_layers_11_self_attn_v_proj_weight_palettized, x = hidden_states_111)[name = tensor("linear_112")]; + tensor var_1724 = const()[name = tensor("op_1724"), val = tensor([1, 2, -1, 64])]; + tensor var_1725 = reshape(shape = var_1724, x = key_states_89)[name = tensor("op_1725")]; + tensor var_1727 = const()[name = tensor("op_1727"), val = tensor([1, 2, -1, 64])]; + tensor var_1728 = reshape(shape = var_1727, x = value_states_89)[name = tensor("op_1728")]; + tensor value_states_91_perm_0 = const()[name = tensor("value_states_91_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_45_interleave_0 = const()[name = tensor("key_45_interleave_0"), val = tensor(false)]; + tensor const_134 = const()[name = tensor("const_134"), val = tensor(1)]; + tensor key_45 = concat(axis = const_134, interleave = key_45_interleave_0, values = var_1725)[name = tensor("key_45")]; + tensor value_45_interleave_0 = const()[name = tensor("value_45_interleave_0"), val = tensor(false)]; + tensor value_states_91 = transpose(perm = value_states_91_perm_0, x = var_1728)[name = tensor("transpose_127")]; + tensor value_45 = concat(axis = var_13, interleave = value_45_interleave_0, values = value_states_91)[name = tensor("value_45")]; + tensor mul_22 = mul(x = var_1716, y = var_11)[name = tensor("mul_22")]; + tensor matmul_22_transpose_y_0 = const()[name = tensor("matmul_22_transpose_y_0"), val = tensor(true)]; + tensor matmul_22_transpose_x_0 = const()[name = tensor("matmul_22_transpose_x_0"), val = tensor(false)]; + tensor transpose_94_perm_0 = const()[name = tensor("transpose_94_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_95_perm_0 = const()[name = tensor("transpose_95_perm_0"), val = tensor([0, 2, -3, -1])]; + tensor transpose_95 = transpose(perm = transpose_95_perm_0, x = key_45)[name = tensor("transpose_125")]; + tensor transpose_94 = transpose(perm = transpose_94_perm_0, x = mul_22)[name = tensor("transpose_126")]; + tensor matmul_22 = matmul(transpose_x = matmul_22_transpose_x_0, transpose_y = matmul_22_transpose_y_0, x = transpose_94, y = transpose_95)[name = tensor("matmul_22")]; + tensor add_22 = add(x = matmul_22, y = reshape_4)[name = tensor("add_22")]; + tensor softmax_22_axis_0 = const()[name = tensor("softmax_22_axis_0"), val = tensor(-1)]; + tensor softmax_22 = softmax(axis = softmax_22_axis_0, x = add_22)[name = tensor("softmax_22")]; + tensor attn_output_89_transpose_x_0 = const()[name = tensor("attn_output_89_transpose_x_0"), val = tensor(false)]; + tensor attn_output_89_transpose_y_0 = const()[name = tensor("attn_output_89_transpose_y_0"), val = tensor(false)]; + tensor attn_output_89 = matmul(transpose_x = attn_output_89_transpose_x_0, transpose_y = attn_output_89_transpose_y_0, x = softmax_22, y = value_45)[name = tensor("attn_output_89")]; + tensor var_1744_perm_0 = const()[name = tensor("op_1744_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1746 = const()[name = tensor("op_1746"), val = tensor([1, 2, -1])]; + tensor var_1744 = transpose(perm = var_1744_perm_0, x = attn_output_89)[name = tensor("transpose_124")]; + tensor var_1747 = reshape(shape = var_1746, x = var_1744)[name = tensor("op_1747")]; + tensor input_273 = linear(bias = decoder_layers_11_self_attn_out_proj_bias, weight = decoder_layers_11_self_attn_out_proj_weight_palettized, x = var_1747)[name = tensor("linear_113")]; + tensor input_275 = add(x = input_269, y = input_273)[name = tensor("input_275")]; + tensor hidden_states_115_axes_0 = const()[name = tensor("hidden_states_115_axes_0"), val = tensor([-1])]; + tensor hidden_states_115 = layer_norm(axes = hidden_states_115_axes_0, beta = decoder_layers_11_encoder_attn_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_11_encoder_attn_layer_norm_weight, x = input_275)[name = tensor("hidden_states_115")]; + tensor var_1771 = linear(bias = decoder_layers_11_encoder_attn_q_proj_bias, weight = decoder_layers_11_encoder_attn_q_proj_weight_palettized, x = hidden_states_115)[name = tensor("linear_114")]; + tensor var_1772 = const()[name = tensor("op_1772"), val = tensor([1, 2, -1, 64])]; + tensor var_1773 = reshape(shape = var_1772, x = var_1771)[name = tensor("op_1773")]; + tensor query_perm_0 = const()[name = tensor("query_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_states_93 = linear(bias = decoder_layers_11_encoder_attn_k_proj_bias, weight = decoder_layers_11_encoder_attn_k_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_115")]; + tensor value_states_93 = linear(bias = decoder_layers_11_encoder_attn_v_proj_bias, weight = decoder_layers_11_encoder_attn_v_proj_weight_palettized, x = encoder_hidden_states)[name = tensor("linear_116")]; + tensor concat_37x = const()[name = tensor("concat_37x"), val = tensor([1, -1, 16, 64])]; + tensor var_1782 = reshape(shape = concat_37x, x = key_states_93)[name = tensor("op_1782")]; + tensor key_states_perm_0 = const()[name = tensor("key_states_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor concat_38x = const()[name = tensor("concat_38x"), val = tensor([1, -1, 16, 64])]; + tensor var_1785 = reshape(shape = concat_38x, x = value_states_93)[name = tensor("op_1785")]; + tensor value_states_perm_0 = const()[name = tensor("value_states_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor key_interleave_0 = const()[name = tensor("key_interleave_0"), val = tensor(false)]; + tensor key_states = transpose(perm = key_states_perm_0, x = var_1782)[name = tensor("transpose_123")]; + tensor key = concat(axis = var_13, interleave = key_interleave_0, values = key_states)[name = tensor("key")]; + tensor value_interleave_0 = const()[name = tensor("value_interleave_0"), val = tensor(false)]; + tensor value_states = transpose(perm = value_states_perm_0, x = var_1785)[name = tensor("transpose_122")]; + tensor value = concat(axis = var_13, interleave = value_interleave_0, values = value_states)[name = tensor("value")]; + tensor var_1795_shape = shape(x = key)[name = tensor("op_1795_shape")]; + tensor gather_25_indices_0 = const()[name = tensor("gather_25_indices_0"), val = tensor(2)]; + tensor gather_25_axis_0 = const()[name = tensor("gather_25_axis_0"), val = tensor(0)]; + tensor gather_25_batch_dims_0 = const()[name = tensor("gather_25_batch_dims_0"), val = tensor(0)]; + tensor gather_25 = gather(axis = gather_25_axis_0, batch_dims = gather_25_batch_dims_0, indices = gather_25_indices_0, x = var_1795_shape)[name = tensor("gather_25")]; + tensor concat_39_values0_0 = const()[name = tensor("concat_39_values0_0"), val = tensor(0)]; + tensor concat_39_values1_0 = const()[name = tensor("concat_39_values1_0"), val = tensor(0)]; + tensor concat_39_values2_0 = const()[name = tensor("concat_39_values2_0"), val = tensor(0)]; + tensor concat_39_axis_0 = const()[name = tensor("concat_39_axis_0"), val = tensor(0)]; + tensor concat_39_interleave_0 = const()[name = tensor("concat_39_interleave_0"), val = tensor(false)]; + tensor concat_39 = concat(axis = concat_39_axis_0, interleave = concat_39_interleave_0, values = (concat_39_values0_0, concat_39_values1_0, concat_39_values2_0, gather_25))[name = tensor("concat_39")]; + tensor attention_mask_begin_0 = const()[name = tensor("attention_mask_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor attention_mask_end_mask_0 = const()[name = tensor("attention_mask_end_mask_0"), val = tensor([true, true, true, false])]; + tensor attention_mask = slice_by_index(begin = attention_mask_begin_0, end = concat_39, end_mask = attention_mask_end_mask_0, x = attention_mask_5)[name = tensor("attention_mask")]; + tensor query = transpose(perm = query_perm_0, x = var_1773)[name = tensor("transpose_121")]; + tensor mul_23 = mul(x = query, y = var_11)[name = tensor("mul_23")]; + tensor matmul_23_transpose_y_0 = const()[name = tensor("matmul_23_transpose_y_0"), val = tensor(true)]; + tensor matmul_23_transpose_x_0 = const()[name = tensor("matmul_23_transpose_x_0"), val = tensor(false)]; + tensor matmul_23 = matmul(transpose_x = matmul_23_transpose_x_0, transpose_y = matmul_23_transpose_y_0, x = mul_23, y = key)[name = tensor("matmul_23")]; + tensor add_23 = add(x = matmul_23, y = attention_mask)[name = tensor("add_23")]; + tensor softmax_23_axis_0 = const()[name = tensor("softmax_23_axis_0"), val = tensor(-1)]; + tensor softmax_23 = softmax(axis = softmax_23_axis_0, x = add_23)[name = tensor("softmax_23")]; + tensor attn_output_93_transpose_x_0 = const()[name = tensor("attn_output_93_transpose_x_0"), val = tensor(false)]; + tensor attn_output_93_transpose_y_0 = const()[name = tensor("attn_output_93_transpose_y_0"), val = tensor(false)]; + tensor attn_output_93 = matmul(transpose_x = attn_output_93_transpose_x_0, transpose_y = attn_output_93_transpose_y_0, x = softmax_23, y = value)[name = tensor("attn_output_93")]; + tensor var_1801_perm_0 = const()[name = tensor("op_1801_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1803 = const()[name = tensor("op_1803"), val = tensor([1, 2, -1])]; + tensor var_1801 = transpose(perm = var_1801_perm_0, x = attn_output_93)[name = tensor("transpose_120")]; + tensor var_1804 = reshape(shape = var_1803, x = var_1801)[name = tensor("op_1804")]; + tensor input_279 = linear(bias = decoder_layers_11_encoder_attn_out_proj_bias, weight = decoder_layers_11_encoder_attn_out_proj_weight_palettized, x = var_1804)[name = tensor("linear_117")]; + tensor input_281 = add(x = input_275, y = input_279)[name = tensor("input_281")]; + tensor input_283_axes_0 = const()[name = tensor("input_283_axes_0"), val = tensor([-1])]; + tensor input_283 = layer_norm(axes = input_283_axes_0, beta = decoder_layers_11_final_layer_norm_bias, epsilon = var_9, gamma = decoder_layers_11_final_layer_norm_weight, x = input_281)[name = tensor("input_283")]; + tensor input_285 = linear(bias = decoder_layers_11_fc1_bias, weight = decoder_layers_11_fc1_weight_palettized, x = input_283)[name = tensor("linear_118")]; + tensor input_287 = relu(x = input_285)[name = tensor("input_287")]; + tensor input_291 = linear(bias = decoder_layers_11_fc2_bias, weight = decoder_layers_11_fc2_weight_palettized, x = input_287)[name = tensor("linear_119")]; + tensor input_293 = add(x = input_281, y = input_291)[name = tensor("input_293")]; + tensor var_1838_axes_0 = const()[name = tensor("op_1838_axes_0"), val = tensor([-1])]; + tensor var_1838 = layer_norm(axes = var_1838_axes_0, beta = decoder_layer_norm_bias, epsilon = var_9, gamma = decoder_layer_norm_weight, x = input_293)[name = tensor("op_1838")]; + tensor var_1898_begin_0 = const()[name = tensor("op_1898_begin_0"), val = tensor([0, -1, 0])]; + tensor var_1898_end_0 = const()[name = tensor("op_1898_end_0"), val = tensor([1, 2, 1024])]; + tensor var_1898_end_mask_0 = const()[name = tensor("op_1898_end_mask_0"), val = tensor([true, true, true])]; + tensor var_1898 = slice_by_index(begin = var_1898_begin_0, end = var_1898_end_0, end_mask = var_1898_end_mask_0, x = var_1838)[name = tensor("op_1898")]; + tensor linear_120_bias_0 = const()[name = tensor("linear_120_bias_0"), val = tensor(BLOBFILE(path = tensor("@model_path/weights/weight.bin"), offset = tensor(465816384)))]; + tensor logits = linear(bias = linear_120_bias_0, weight = decoder_embed_tokens_weight_palettized, x = var_1898)[name = tensor("linear_120")]; + tensor var_1908_axis_0 = const()[name = tensor("op_1908_axis_0"), val = tensor(0)]; + tensor transpose_96_perm_0 = const()[name = tensor("transpose_96_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor transpose_97_perm_0 = const()[name = tensor("transpose_97_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor transpose_98_perm_0 = const()[name = tensor("transpose_98_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor transpose_99_perm_0 = const()[name = tensor("transpose_99_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor transpose_100_perm_0 = const()[name = tensor("transpose_100_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor transpose_101_perm_0 = const()[name = tensor("transpose_101_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor transpose_102_perm_0 = const()[name = tensor("transpose_102_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor transpose_103_perm_0 = const()[name = tensor("transpose_103_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor transpose_104_perm_0 = const()[name = tensor("transpose_104_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor transpose_105_perm_0 = const()[name = tensor("transpose_105_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor transpose_106_perm_0 = const()[name = tensor("transpose_106_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor transpose_107_perm_0 = const()[name = tensor("transpose_107_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor transpose_107 = transpose(perm = transpose_107_perm_0, x = key_45)[name = tensor("transpose_108")]; + tensor transpose_106 = transpose(perm = transpose_106_perm_0, x = key_41)[name = tensor("transpose_109")]; + tensor transpose_105 = transpose(perm = transpose_105_perm_0, x = key_37)[name = tensor("transpose_110")]; + tensor transpose_104 = transpose(perm = transpose_104_perm_0, x = key_33)[name = tensor("transpose_111")]; + tensor transpose_103 = transpose(perm = transpose_103_perm_0, x = key_29)[name = tensor("transpose_112")]; + tensor transpose_102 = transpose(perm = transpose_102_perm_0, x = key_25)[name = tensor("transpose_113")]; + tensor transpose_101 = transpose(perm = transpose_101_perm_0, x = key_21)[name = tensor("transpose_114")]; + tensor transpose_100 = transpose(perm = transpose_100_perm_0, x = key_17)[name = tensor("transpose_115")]; + tensor transpose_99 = transpose(perm = transpose_99_perm_0, x = key_13)[name = tensor("transpose_116")]; + tensor transpose_98 = transpose(perm = transpose_98_perm_0, x = key_9)[name = tensor("transpose_117")]; + tensor transpose_97 = transpose(perm = transpose_97_perm_0, x = key_5)[name = tensor("transpose_118")]; + tensor transpose_96 = transpose(perm = transpose_96_perm_0, x = key_1)[name = tensor("transpose_119")]; + tensor past_self_key = stack(axis = var_1908_axis_0, values = (transpose_96, transpose_97, transpose_98, transpose_99, transpose_100, transpose_101, transpose_102, transpose_103, transpose_104, transpose_105, transpose_106, transpose_107))[name = tensor("op_1908")]; + tensor var_1911_axis_0 = const()[name = tensor("op_1911_axis_0"), val = tensor(0)]; + tensor past_self_value = stack(axis = var_1911_axis_0, values = (value_1, value_5, value_9, value_13, value_17, value_21, value_25, value_29, value_33, value_37, value_41, value_45))[name = tensor("op_1911")]; + tensor var_1914_axis_0 = const()[name = tensor("op_1914_axis_0"), val = tensor(0)]; + tensor past_cross_key = stack(axis = var_1914_axis_0, values = (key_3, key_7, key_11, key_15, key_19, key_23, key_27, key_31, key_35, key_39, key_43, key))[name = tensor("op_1914")]; + tensor var_1917_axis_0 = const()[name = tensor("op_1917_axis_0"), val = tensor(0)]; + tensor past_cross_value = stack(axis = var_1917_axis_0, values = (value_3, value_7, value_11, value_15, value_19, value_23, value_27, value_31, value_35, value_39, value_43, value))[name = tensor("op_1917")]; + } -> (logits, past_self_key, past_self_value, past_cross_key, past_cross_value); +} \ No newline at end of file