diff --git "a/r2t2_FFN_PF_lut8_chunk_01of02.mlmodelc/model.mil" "b/r2t2_FFN_PF_lut8_chunk_01of02.mlmodelc/model.mil" new file mode 100644--- /dev/null +++ "b/r2t2_FFN_PF_lut8_chunk_01of02.mlmodelc/model.mil" @@ -0,0 +1,6947 @@ +program(1.3) +[buildInfo = dict({{"coremlc-component-MIL", "3600.16.1"}, {"coremlc-version", "3600.25.1"}})] +{ + func infer(tensor causal_mask, tensor current_pos, tensor hidden_states, state> model_model_kv_cache_0, tensor position_ids) { + tensor model_model_layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4194432))))[name = string("model_model_layers_0_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_0_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4325568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6422784))))[name = string("model_model_layers_0_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_0_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6488384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8585600))))[name = string("model_model_layers_0_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_0_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8651200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21234176))))[name = string("model_model_layers_0_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_0_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21627456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34210432))))[name = string("model_model_layers_0_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_0_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34603712))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47186688))))[name = string("model_model_layers_0_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_1_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47317824))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51512192))))[name = string("model_model_layers_1_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_1_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51643328))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53740544))))[name = string("model_model_layers_1_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_1_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53806144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55903360))))[name = string("model_model_layers_1_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_1_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55968960))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68551936))))[name = string("model_model_layers_1_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_1_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68945216))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(81528192))))[name = string("model_model_layers_1_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_1_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(81921472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(94504448))))[name = string("model_model_layers_1_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(94635584))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(98829952))))[name = string("model_model_layers_2_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_2_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(98961088))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101058304))))[name = string("model_model_layers_2_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_2_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101123904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103221120))))[name = string("model_model_layers_2_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_2_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103286720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115869696))))[name = string("model_model_layers_2_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_2_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116262976))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(128845952))))[name = string("model_model_layers_2_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_2_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(129239232))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141822208))))[name = string("model_model_layers_2_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_3_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141953344))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146147712))))[name = string("model_model_layers_3_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_3_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146278848))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148376064))))[name = string("model_model_layers_3_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_3_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148441664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(150538880))))[name = string("model_model_layers_3_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_3_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(150604480))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163187456))))[name = string("model_model_layers_3_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_3_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163580736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176163712))))[name = string("model_model_layers_3_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_3_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176556992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(189139968))))[name = string("model_model_layers_3_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_4_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(189271104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193465472))))[name = string("model_model_layers_4_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_4_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193596608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(195693824))))[name = string("model_model_layers_4_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_4_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(195759424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(197856640))))[name = string("model_model_layers_4_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_4_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(197922240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210505216))))[name = string("model_model_layers_4_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_4_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210898496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223481472))))[name = string("model_model_layers_4_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_4_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223874752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236457728))))[name = string("model_model_layers_4_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_5_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236588864))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240783232))))[name = string("model_model_layers_5_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_5_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240914368))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(243011584))))[name = string("model_model_layers_5_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_5_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(243077184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245174400))))[name = string("model_model_layers_5_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_5_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245240000))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(257822976))))[name = string("model_model_layers_5_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_5_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258216256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(270799232))))[name = string("model_model_layers_5_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_5_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(271192512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283775488))))[name = string("model_model_layers_5_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_6_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283906624))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288100992))))[name = string("model_model_layers_6_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_6_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288232128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290329344))))[name = string("model_model_layers_6_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_6_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290394944))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292492160))))[name = string("model_model_layers_6_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_6_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292557760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305140736))))[name = string("model_model_layers_6_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_6_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305534016))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318116992))))[name = string("model_model_layers_6_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_6_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318510272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331093248))))[name = string("model_model_layers_6_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_7_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331224384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335418752))))[name = string("model_model_layers_7_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_7_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335549888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(337647104))))[name = string("model_model_layers_7_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_7_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(337712704))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339809920))))[name = string("model_model_layers_7_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_7_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339875520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352458496))))[name = string("model_model_layers_7_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_7_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352851776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(365434752))))[name = string("model_model_layers_7_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_7_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(365828032))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(378411008))))[name = string("model_model_layers_7_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_8_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(378542144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(382736512))))[name = string("model_model_layers_8_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_8_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(382867648))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(384964864))))[name = string("model_model_layers_8_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_8_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385030464))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(387127680))))[name = string("model_model_layers_8_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_8_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(387193280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399776256))))[name = string("model_model_layers_8_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_8_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400169536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(412752512))))[name = string("model_model_layers_8_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_8_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(413145792))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(425728768))))[name = string("model_model_layers_8_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_9_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(425859904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(430054272))))[name = string("model_model_layers_9_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_9_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(430185408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432282624))))[name = string("model_model_layers_9_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_9_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432348224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(434445440))))[name = string("model_model_layers_9_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_9_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(434511040))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(447094016))))[name = string("model_model_layers_9_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_9_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(447487296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(460070272))))[name = string("model_model_layers_9_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_9_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(460463552))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(473046528))))[name = string("model_model_layers_9_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_10_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(473177664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(477372032))))[name = string("model_model_layers_10_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_10_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(477503168))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479600384))))[name = string("model_model_layers_10_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_10_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479665984))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(481763200))))[name = string("model_model_layers_10_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_10_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(481828800))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(494411776))))[name = string("model_model_layers_10_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_10_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(494805056))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(507388032))))[name = string("model_model_layers_10_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_10_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(507781312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(520364288))))[name = string("model_model_layers_10_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_11_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(520495424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524689792))))[name = string("model_model_layers_11_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_11_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524820928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526918144))))[name = string("model_model_layers_11_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_11_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526983744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(529080960))))[name = string("model_model_layers_11_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_11_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(529146560))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(541729536))))[name = string("model_model_layers_11_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_11_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(542122816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554705792))))[name = string("model_model_layers_11_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_11_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(555099072))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(567682048))))[name = string("model_model_layers_11_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_12_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(567813184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(572007552))))[name = string("model_model_layers_12_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_12_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(572138688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(574235904))))[name = string("model_model_layers_12_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_12_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(574301504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(576398720))))[name = string("model_model_layers_12_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_12_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(576464320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(589047296))))[name = string("model_model_layers_12_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_12_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(589440576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(602023552))))[name = string("model_model_layers_12_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_12_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(602416832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(614999808))))[name = string("model_model_layers_12_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_13_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(615130944))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(619325312))))[name = string("model_model_layers_13_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_13_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(619456448))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(621553664))))[name = string("model_model_layers_13_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_13_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(621619264))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(623716480))))[name = string("model_model_layers_13_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_13_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(623782080))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(636365056))))[name = string("model_model_layers_13_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_13_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(636758336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(649341312))))[name = string("model_model_layers_13_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_13_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(649734592))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(662317568))))[name = string("model_model_layers_13_mlp_down_proj_weight_palettized")]; + int32 var_758_batch_dims_0 = const()[name = string("op_758_batch_dims_0"), val = int32(0)]; + bool var_758_validate_indices_0 = const()[name = string("op_758_validate_indices_0"), val = bool(false)]; + tensor var_750_to_fp16 = const()[name = string("op_750_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(662448704)))]; + string current_pos_to_int16_dtype_0 = const()[name = string("current_pos_to_int16_dtype_0"), val = string("int16")]; + string cast_118_dtype_0 = const()[name = string("cast_118_dtype_0"), val = string("int32")]; + int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; + tensor current_pos_to_int16 = cast(dtype = current_pos_to_int16_dtype_0, x = current_pos)[name = string("cast_5")]; + tensor cast_118 = cast(dtype = cast_118_dtype_0, x = current_pos_to_int16)[name = string("cast_4")]; + tensor greater_equal_0 = greater_equal(x = cast_118, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; + int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(2048)]; + tensor add_0 = add(x = cast_118, y = slice_by_index_0)[name = string("add_0")]; + tensor select_0 = select(a = cast_118, b = add_0, cond = greater_equal_0)[name = string("select_0")]; + string select_0_to_int16_dtype_0 = const()[name = string("select_0_to_int16_dtype_0"), val = string("int16")]; + string cast_0_dtype_0 = const()[name = string("cast_0_dtype_0"), val = string("int32")]; + int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; + tensor select_0_to_int16 = cast(dtype = select_0_to_int16_dtype_0, x = select_0)[name = string("cast_3")]; + tensor cast_0 = cast(dtype = cast_0_dtype_0, x = select_0_to_int16)[name = string("cast_2")]; + tensor greater_equal_0_1 = greater_equal(x = cast_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; + int32 slice_by_index_0_1 = const()[name = string("slice_by_index_0_1"), val = int32(2048)]; + tensor add_0_1 = add(x = cast_0, y = slice_by_index_0_1)[name = string("add_0_1")]; + tensor select_0_1 = select(a = cast_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; + int32 op_758_cast_fp16_cast_uint16_cast_uint16_axis_0 = const()[name = string("op_758_cast_fp16_cast_uint16_cast_uint16_axis_0"), val = int32(1)]; + tensor op_758_cast_fp16_cast_uint16_cast_uint16 = gather(axis = op_758_cast_fp16_cast_uint16_cast_uint16_axis_0, batch_dims = var_758_batch_dims_0, indices = select_0_1, validate_indices = var_758_validate_indices_0, x = var_750_to_fp16)[name = string("op_758_cast_fp16_cast_uint16_cast_uint16")]; + tensor var_763 = const()[name = string("op_763"), val = tensor([1, 1, 1, -1])]; + tensor sin_1_cast_fp16 = reshape(shape = var_763, x = op_758_cast_fp16_cast_uint16_cast_uint16)[name = string("sin_1_cast_fp16")]; + int32 var_773_axis_0 = const()[name = string("op_773_axis_0"), val = int32(1)]; + int32 var_773_batch_dims_0 = const()[name = string("op_773_batch_dims_0"), val = int32(0)]; + bool var_773_validate_indices_0 = const()[name = string("op_773_validate_indices_0"), val = bool(false)]; + tensor var_765_to_fp16 = const()[name = string("op_765_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(662973056)))]; + string current_pos_to_uint16_dtype_0 = const()[name = string("current_pos_to_uint16_dtype_0"), val = string("uint16")]; + tensor current_pos_to_uint16 = cast(dtype = current_pos_to_uint16_dtype_0, x = current_pos)[name = string("cast_1")]; + tensor var_773_cast_fp16_cast_uint16 = gather(axis = var_773_axis_0, batch_dims = var_773_batch_dims_0, indices = current_pos_to_uint16, validate_indices = var_773_validate_indices_0, x = var_765_to_fp16)[name = string("op_773_cast_fp16_cast_uint16")]; + tensor var_778 = const()[name = string("op_778"), val = tensor([1, 1, 1, -1])]; + tensor cos_1_cast_fp16 = reshape(shape = var_778, x = var_773_cast_fp16_cast_uint16)[name = string("cos_1_cast_fp16")]; + int32 var_799 = const()[name = string("op_799"), val = int32(-1)]; + fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_801_cast_fp16 = mul(x = hidden_states, y = const_0_promoted_to_fp16)[name = string("op_801_cast_fp16")]; + bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; + tensor input_1_cast_fp16 = concat(axis = var_799, interleave = input_1_interleave_0, values = (hidden_states, var_801_cast_fp16))[name = string("input_1_cast_fp16")]; + tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; + fp16 var_796_to_fp16 = const()[name = string("op_796_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_796_to_fp16, x = input_1_cast_fp16)[name = string("normed_1_cast_fp16")]; + tensor normed_3_begin_0 = const()[name = string("normed_3_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_3_end_0 = const()[name = string("normed_3_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_3_end_mask_0 = const()[name = string("normed_3_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_3_cast_fp16 = slice_by_index(begin = normed_3_begin_0, end = normed_3_end_0, end_mask = normed_3_end_mask_0, x = normed_1_cast_fp16)[name = string("normed_3_cast_fp16")]; + tensor const_3_promoted_to_fp16 = const()[name = string("const_3_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(663497408)))]; + tensor hidden_states_3_cast_fp16 = mul(x = normed_3_cast_fp16, y = const_3_promoted_to_fp16)[name = string("hidden_states_3_cast_fp16")]; + tensor var_818 = const()[name = string("op_818"), val = tensor([0, 2, 1])]; + tensor var_821_axes_0 = const()[name = string("op_821_axes_0"), val = tensor([2])]; + tensor var_819_cast_fp16 = transpose(perm = var_818, x = hidden_states_3_cast_fp16)[name = string("transpose_83")]; + tensor var_821_cast_fp16 = expand_dims(axes = var_821_axes_0, x = var_819_cast_fp16)[name = string("op_821_cast_fp16")]; + string var_837_pad_type_0 = const()[name = string("op_837_pad_type_0"), val = string("valid")]; + tensor var_837_strides_0 = const()[name = string("op_837_strides_0"), val = tensor([1, 1])]; + tensor var_837_pad_0 = const()[name = string("op_837_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_837_dilations_0 = const()[name = string("op_837_dilations_0"), val = tensor([1, 1])]; + int32 var_837_groups_0 = const()[name = string("op_837_groups_0"), val = int32(1)]; + tensor var_837 = conv(dilations = var_837_dilations_0, groups = var_837_groups_0, pad = var_837_pad_0, pad_type = var_837_pad_type_0, strides = var_837_strides_0, weight = model_model_layers_0_self_attn_q_proj_weight_palettized, x = var_821_cast_fp16)[name = string("op_837")]; + tensor var_842 = const()[name = string("op_842"), val = tensor([1, 16, 1, 128])]; + tensor var_843 = reshape(shape = var_842, x = var_837)[name = string("op_843")]; + string var_859_pad_type_0 = const()[name = string("op_859_pad_type_0"), val = string("valid")]; + tensor var_859_strides_0 = const()[name = string("op_859_strides_0"), val = tensor([1, 1])]; + tensor var_859_pad_0 = const()[name = string("op_859_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_859_dilations_0 = const()[name = string("op_859_dilations_0"), val = tensor([1, 1])]; + int32 var_859_groups_0 = const()[name = string("op_859_groups_0"), val = int32(1)]; + tensor var_859 = conv(dilations = var_859_dilations_0, groups = var_859_groups_0, pad = var_859_pad_0, pad_type = var_859_pad_type_0, strides = var_859_strides_0, weight = model_model_layers_0_self_attn_k_proj_weight_palettized, x = var_821_cast_fp16)[name = string("op_859")]; + tensor var_864 = const()[name = string("op_864"), val = tensor([1, 8, 1, 128])]; + tensor var_865 = reshape(shape = var_864, x = var_859)[name = string("op_865")]; + string var_881_pad_type_0 = const()[name = string("op_881_pad_type_0"), val = string("valid")]; + tensor var_881_strides_0 = const()[name = string("op_881_strides_0"), val = tensor([1, 1])]; + tensor var_881_pad_0 = const()[name = string("op_881_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_881_dilations_0 = const()[name = string("op_881_dilations_0"), val = tensor([1, 1])]; + int32 var_881_groups_0 = const()[name = string("op_881_groups_0"), val = int32(1)]; + tensor var_881 = conv(dilations = var_881_dilations_0, groups = var_881_groups_0, pad = var_881_pad_0, pad_type = var_881_pad_type_0, strides = var_881_strides_0, weight = model_model_layers_0_self_attn_v_proj_weight_palettized, x = var_821_cast_fp16)[name = string("op_881")]; + tensor var_886 = const()[name = string("op_886"), val = tensor([1, 8, 1, 128])]; + tensor var_887 = reshape(shape = var_886, x = var_881)[name = string("op_887")]; + int32 var_902 = const()[name = string("op_902"), val = int32(-1)]; + fp16 const_4_promoted = const()[name = string("const_4_promoted"), val = fp16(-0x1p+0)]; + tensor var_904 = mul(x = var_843, y = const_4_promoted)[name = string("op_904")]; + bool input_5_interleave_0 = const()[name = string("input_5_interleave_0"), val = bool(false)]; + tensor input_5 = concat(axis = var_902, interleave = input_5_interleave_0, values = (var_843, var_904))[name = string("input_5")]; + tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; + fp16 var_899_to_fp16 = const()[name = string("op_899_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_899_to_fp16, x = input_5)[name = string("normed_5_cast_fp16")]; + tensor normed_7_begin_0 = const()[name = string("normed_7_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_7_end_0 = const()[name = string("normed_7_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_7_end_mask_0 = const()[name = string("normed_7_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_7 = slice_by_index(begin = normed_7_begin_0, end = normed_7_end_0, end_mask = normed_7_end_mask_0, x = normed_5_cast_fp16)[name = string("normed_7")]; + tensor const_7 = const()[name = string("const_7"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(663501568)))]; + tensor q_1 = mul(x = normed_7, y = const_7)[name = string("q_1")]; + int32 var_927 = const()[name = string("op_927"), val = int32(-1)]; + fp16 const_8_promoted = const()[name = string("const_8_promoted"), val = fp16(-0x1p+0)]; + tensor var_929 = mul(x = var_865, y = const_8_promoted)[name = string("op_929")]; + bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; + tensor input_7 = concat(axis = var_927, interleave = input_7_interleave_0, values = (var_865, var_929))[name = string("input_7")]; + tensor normed_9_axes_0 = const()[name = string("normed_9_axes_0"), val = tensor([-1])]; + fp16 var_924_to_fp16 = const()[name = string("op_924_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_9_cast_fp16 = layer_norm(axes = normed_9_axes_0, epsilon = var_924_to_fp16, x = input_7)[name = string("normed_9_cast_fp16")]; + tensor normed_11_begin_0 = const()[name = string("normed_11_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_11_end_0 = const()[name = string("normed_11_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_11_end_mask_0 = const()[name = string("normed_11_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_11 = slice_by_index(begin = normed_11_begin_0, end = normed_11_end_0, end_mask = normed_11_end_mask_0, x = normed_9_cast_fp16)[name = string("normed_11")]; + tensor const_11 = const()[name = string("const_11"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(663501888)))]; + tensor k_1 = mul(x = normed_11, y = const_11)[name = string("k_1")]; + tensor var_943 = mul(x = q_1, y = cos_1_cast_fp16)[name = string("op_943")]; + tensor x1_1_begin_0 = const()[name = string("x1_1_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_1_end_0 = const()[name = string("x1_1_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_1_end_mask_0 = const()[name = string("x1_1_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_1 = slice_by_index(begin = x1_1_begin_0, end = x1_1_end_0, end_mask = x1_1_end_mask_0, x = q_1)[name = string("x1_1")]; + tensor x2_1_begin_0 = const()[name = string("x2_1_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_1_end_0 = const()[name = string("x2_1_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_1_end_mask_0 = const()[name = string("x2_1_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_1 = slice_by_index(begin = x2_1_begin_0, end = x2_1_end_0, end_mask = x2_1_end_mask_0, x = q_1)[name = string("x2_1")]; + fp16 const_14_promoted = const()[name = string("const_14_promoted"), val = fp16(-0x1p+0)]; + tensor var_964 = mul(x = x2_1, y = const_14_promoted)[name = string("op_964")]; + int32 var_966 = const()[name = string("op_966"), val = int32(-1)]; + bool var_967_interleave_0 = const()[name = string("op_967_interleave_0"), val = bool(false)]; + tensor var_967 = concat(axis = var_966, interleave = var_967_interleave_0, values = (var_964, x1_1))[name = string("op_967")]; + tensor var_968 = mul(x = var_967, y = sin_1_cast_fp16)[name = string("op_968")]; + tensor query_states_1 = add(x = var_943, y = var_968)[name = string("query_states_1")]; + tensor var_971 = mul(x = k_1, y = cos_1_cast_fp16)[name = string("op_971")]; + tensor x1_3_begin_0 = const()[name = string("x1_3_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_3_end_0 = const()[name = string("x1_3_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_3_end_mask_0 = const()[name = string("x1_3_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_3 = slice_by_index(begin = x1_3_begin_0, end = x1_3_end_0, end_mask = x1_3_end_mask_0, x = k_1)[name = string("x1_3")]; + tensor x2_3_begin_0 = const()[name = string("x2_3_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_3_end_0 = const()[name = string("x2_3_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_3_end_mask_0 = const()[name = string("x2_3_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_3 = slice_by_index(begin = x2_3_begin_0, end = x2_3_end_0, end_mask = x2_3_end_mask_0, x = k_1)[name = string("x2_3")]; + fp16 const_17_promoted = const()[name = string("const_17_promoted"), val = fp16(-0x1p+0)]; + tensor var_992 = mul(x = x2_3, y = const_17_promoted)[name = string("op_992")]; + int32 var_994 = const()[name = string("op_994"), val = int32(-1)]; + bool var_995_interleave_0 = const()[name = string("op_995_interleave_0"), val = bool(false)]; + tensor var_995 = concat(axis = var_994, interleave = var_995_interleave_0, values = (var_992, x1_3))[name = string("op_995")]; + tensor var_996 = mul(x = var_995, y = sin_1_cast_fp16)[name = string("op_996")]; + tensor key_states_1 = add(x = var_971, y = var_996)[name = string("key_states_1")]; + int32 var_1000 = const()[name = string("op_1000"), val = int32(1)]; + tensor var_1001 = add(x = current_pos, y = var_1000)[name = string("op_1001")]; + tensor read_state_0 = read_state(input = model_model_kv_cache_0)[name = string("read_state_0")]; + tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([0])]; + tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; + tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; + tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([1])]; + int32 concat_2_axis_0 = const()[name = string("concat_2_axis_0"), val = int32(0)]; + bool concat_2_interleave_0 = const()[name = string("concat_2_interleave_0"), val = bool(false)]; + tensor concat_2 = concat(axis = concat_2_axis_0, interleave = concat_2_interleave_0, values = (expand_dims_0, expand_dims_1, current_pos, expand_dims_3))[name = string("concat_2")]; + tensor concat_3_values1_0 = const()[name = string("concat_3_values1_0"), val = tensor([0])]; + tensor concat_3_values3_0 = const()[name = string("concat_3_values3_0"), val = tensor([0])]; + int32 concat_3_axis_0 = const()[name = string("concat_3_axis_0"), val = int32(0)]; + bool concat_3_interleave_0 = const()[name = string("concat_3_interleave_0"), val = bool(false)]; + tensor concat_3 = concat(axis = concat_3_axis_0, interleave = concat_3_interleave_0, values = (expand_dims_4, concat_3_values1_0, var_1001, concat_3_values3_0))[name = string("concat_3")]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_2, begin_mask = model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0, end = concat_3, end_mask = model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_1_stride_0, update = key_states_1, x = read_state_0)[name = string("model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_56_write_state")]; + tensor coreml_update_state_28 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_56")]; + tensor expand_dims_6 = const()[name = string("expand_dims_6"), val = tensor([28])]; + tensor expand_dims_7 = const()[name = string("expand_dims_7"), val = tensor([0])]; + tensor expand_dims_9 = const()[name = string("expand_dims_9"), val = tensor([0])]; + tensor expand_dims_10 = const()[name = string("expand_dims_10"), val = tensor([29])]; + int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; + bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; + tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_6, expand_dims_7, current_pos, expand_dims_9))[name = string("concat_6")]; + tensor concat_7_values1_0 = const()[name = string("concat_7_values1_0"), val = tensor([0])]; + tensor concat_7_values3_0 = const()[name = string("concat_7_values3_0"), val = tensor([0])]; + int32 concat_7_axis_0 = const()[name = string("concat_7_axis_0"), val = int32(0)]; + bool concat_7_interleave_0 = const()[name = string("concat_7_interleave_0"), val = bool(false)]; + tensor concat_7 = concat(axis = concat_7_axis_0, interleave = concat_7_interleave_0, values = (expand_dims_10, concat_7_values1_0, var_1001, concat_7_values3_0))[name = string("concat_7")]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_6, begin_mask = model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0, end = concat_7, end_mask = model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_2_stride_0, update = var_887, x = coreml_update_state_28)[name = string("model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_57_write_state")]; + tensor coreml_update_state_29 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_57")]; + tensor var_1051_begin_0 = const()[name = string("op_1051_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1051_end_0 = const()[name = string("op_1051_end_0"), val = tensor([1, 8, 1024, 128])]; + tensor var_1051_end_mask_0 = const()[name = string("op_1051_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1051_cast_fp16 = slice_by_index(begin = var_1051_begin_0, end = var_1051_end_0, end_mask = var_1051_end_mask_0, x = coreml_update_state_29)[name = string("op_1051_cast_fp16")]; + tensor K_layer_cache_1_axes_0 = const()[name = string("K_layer_cache_1_axes_0"), val = tensor([0])]; + tensor K_layer_cache_1_cast_fp16 = squeeze(axes = K_layer_cache_1_axes_0, x = var_1051_cast_fp16)[name = string("K_layer_cache_1_cast_fp16")]; + tensor var_1058_begin_0 = const()[name = string("op_1058_begin_0"), val = tensor([28, 0, 0, 0])]; + tensor var_1058_end_0 = const()[name = string("op_1058_end_0"), val = tensor([29, 8, 1024, 128])]; + tensor var_1058_end_mask_0 = const()[name = string("op_1058_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1058_cast_fp16 = slice_by_index(begin = var_1058_begin_0, end = var_1058_end_0, end_mask = var_1058_end_mask_0, x = coreml_update_state_29)[name = string("op_1058_cast_fp16")]; + tensor V_layer_cache_1_axes_0 = const()[name = string("V_layer_cache_1_axes_0"), val = tensor([0])]; + tensor V_layer_cache_1_cast_fp16 = squeeze(axes = V_layer_cache_1_axes_0, x = var_1058_cast_fp16)[name = string("V_layer_cache_1_cast_fp16")]; + tensor x_3_axes_0 = const()[name = string("x_3_axes_0"), val = tensor([1])]; + tensor x_3_cast_fp16 = expand_dims(axes = x_3_axes_0, x = K_layer_cache_1_cast_fp16)[name = string("x_3_cast_fp16")]; + tensor var_1095 = const()[name = string("op_1095"), val = tensor([1, 2, 1, 1])]; + tensor x_5_cast_fp16 = tile(reps = var_1095, x = x_3_cast_fp16)[name = string("x_5_cast_fp16")]; + tensor var_1107 = const()[name = string("op_1107"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_3_cast_fp16 = reshape(shape = var_1107, x = x_5_cast_fp16)[name = string("key_states_3_cast_fp16")]; + tensor x_9_axes_0 = const()[name = string("x_9_axes_0"), val = tensor([1])]; + tensor x_9_cast_fp16 = expand_dims(axes = x_9_axes_0, x = V_layer_cache_1_cast_fp16)[name = string("x_9_cast_fp16")]; + tensor var_1115 = const()[name = string("op_1115"), val = tensor([1, 2, 1, 1])]; + tensor x_11_cast_fp16 = tile(reps = var_1115, x = x_9_cast_fp16)[name = string("x_11_cast_fp16")]; + tensor var_1127 = const()[name = string("op_1127"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_3_cast_fp16 = reshape(shape = var_1127, x = x_11_cast_fp16)[name = string("value_states_3_cast_fp16")]; + bool var_1142_transpose_x_1 = const()[name = string("op_1142_transpose_x_1"), val = bool(false)]; + bool var_1142_transpose_y_1 = const()[name = string("op_1142_transpose_y_1"), val = bool(true)]; + tensor var_1142 = matmul(transpose_x = var_1142_transpose_x_1, transpose_y = var_1142_transpose_y_1, x = query_states_1, y = key_states_3_cast_fp16)[name = string("op_1142")]; + fp16 var_1143_to_fp16 = const()[name = string("op_1143_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_1_cast_fp16 = mul(x = var_1142, y = var_1143_to_fp16)[name = string("attn_weights_1_cast_fp16")]; + tensor attn_weights_3_cast_fp16 = add(x = attn_weights_1_cast_fp16, y = causal_mask)[name = string("attn_weights_3_cast_fp16")]; + int32 var_1178 = const()[name = string("op_1178"), val = int32(-1)]; + tensor attn_weights_5_cast_fp16 = softmax(axis = var_1178, x = attn_weights_3_cast_fp16)[name = string("attn_weights_5_cast_fp16")]; + bool attn_output_1_transpose_x_0 = const()[name = string("attn_output_1_transpose_x_0"), val = bool(false)]; + bool attn_output_1_transpose_y_0 = const()[name = string("attn_output_1_transpose_y_0"), val = bool(false)]; + tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_0, transpose_y = attn_output_1_transpose_y_0, x = attn_weights_5_cast_fp16, y = value_states_3_cast_fp16)[name = string("attn_output_1_cast_fp16")]; + tensor var_1189_perm_0 = const()[name = string("op_1189_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1193 = const()[name = string("op_1193"), val = tensor([1, 1, 2048])]; + tensor var_1189_cast_fp16 = transpose(perm = var_1189_perm_0, x = attn_output_1_cast_fp16)[name = string("transpose_82")]; + tensor attn_output_5_cast_fp16 = reshape(shape = var_1193, x = var_1189_cast_fp16)[name = string("attn_output_5_cast_fp16")]; + tensor var_1198 = const()[name = string("op_1198"), val = tensor([0, 2, 1])]; + string var_1214_pad_type_0 = const()[name = string("op_1214_pad_type_0"), val = string("valid")]; + int32 var_1214_groups_0 = const()[name = string("op_1214_groups_0"), val = int32(1)]; + tensor var_1214_strides_0 = const()[name = string("op_1214_strides_0"), val = tensor([1])]; + tensor var_1214_pad_0 = const()[name = string("op_1214_pad_0"), val = tensor([0, 0])]; + tensor var_1214_dilations_0 = const()[name = string("op_1214_dilations_0"), val = tensor([1])]; + tensor squeeze_0_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(663502208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667696576))))[name = string("squeeze_0_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_1199_cast_fp16 = transpose(perm = var_1198, x = attn_output_5_cast_fp16)[name = string("transpose_81")]; + tensor var_1214_cast_fp16 = conv(dilations = var_1214_dilations_0, groups = var_1214_groups_0, pad = var_1214_pad_0, pad_type = var_1214_pad_type_0, strides = var_1214_strides_0, weight = squeeze_0_cast_fp16_to_fp32_to_fp16_palettized, x = var_1199_cast_fp16)[name = string("op_1214_cast_fp16")]; + tensor var_1218 = const()[name = string("op_1218"), val = tensor([0, 2, 1])]; + tensor attn_output_9_cast_fp16 = transpose(perm = var_1218, x = var_1214_cast_fp16)[name = string("transpose_80")]; + tensor hidden_states_9_cast_fp16 = add(x = hidden_states, y = attn_output_9_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; + int32 var_1231 = const()[name = string("op_1231"), val = int32(-1)]; + fp16 const_26_promoted_to_fp16 = const()[name = string("const_26_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1233_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_26_promoted_to_fp16)[name = string("op_1233_cast_fp16")]; + bool input_11_interleave_0 = const()[name = string("input_11_interleave_0"), val = bool(false)]; + tensor input_11_cast_fp16 = concat(axis = var_1231, interleave = input_11_interleave_0, values = (hidden_states_9_cast_fp16, var_1233_cast_fp16))[name = string("input_11_cast_fp16")]; + tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; + fp16 var_1228_to_fp16 = const()[name = string("op_1228_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_1228_to_fp16, x = input_11_cast_fp16)[name = string("normed_13_cast_fp16")]; + tensor normed_15_begin_0 = const()[name = string("normed_15_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_15_end_0 = const()[name = string("normed_15_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_15_end_mask_0 = const()[name = string("normed_15_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_15_cast_fp16 = slice_by_index(begin = normed_15_begin_0, end = normed_15_end_0, end_mask = normed_15_end_mask_0, x = normed_13_cast_fp16)[name = string("normed_15_cast_fp16")]; + tensor const_29_promoted_to_fp16 = const()[name = string("const_29_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667827712)))]; + tensor x_13_cast_fp16 = mul(x = normed_15_cast_fp16, y = const_29_promoted_to_fp16)[name = string("x_13_cast_fp16")]; + tensor var_1258 = const()[name = string("op_1258"), val = tensor([0, 2, 1])]; + tensor input_13_axes_0 = const()[name = string("input_13_axes_0"), val = tensor([2])]; + tensor var_1259 = transpose(perm = var_1258, x = x_13_cast_fp16)[name = string("transpose_79")]; + tensor input_13 = expand_dims(axes = input_13_axes_0, x = var_1259)[name = string("input_13")]; + string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; + tensor input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor([1, 1])]; + tensor input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor([1, 1])]; + int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; + tensor input_15 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = model_model_layers_0_mlp_gate_proj_weight_palettized, x = input_13)[name = string("input_15")]; + string b_1_pad_type_0 = const()[name = string("b_1_pad_type_0"), val = string("valid")]; + tensor b_1_strides_0 = const()[name = string("b_1_strides_0"), val = tensor([1, 1])]; + tensor b_1_pad_0 = const()[name = string("b_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_1_dilations_0 = const()[name = string("b_1_dilations_0"), val = tensor([1, 1])]; + int32 b_1_groups_0 = const()[name = string("b_1_groups_0"), val = int32(1)]; + tensor b_1 = conv(dilations = b_1_dilations_0, groups = b_1_groups_0, pad = b_1_pad_0, pad_type = b_1_pad_type_0, strides = b_1_strides_0, weight = model_model_layers_0_mlp_up_proj_weight_palettized, x = input_13)[name = string("b_1")]; + tensor c_1 = silu(x = input_15)[name = string("c_1")]; + tensor input_17 = mul(x = c_1, y = b_1)[name = string("input_17")]; + string e_1_pad_type_0 = const()[name = string("e_1_pad_type_0"), val = string("valid")]; + tensor e_1_strides_0 = const()[name = string("e_1_strides_0"), val = tensor([1, 1])]; + tensor e_1_pad_0 = const()[name = string("e_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_1_dilations_0 = const()[name = string("e_1_dilations_0"), val = tensor([1, 1])]; + int32 e_1_groups_0 = const()[name = string("e_1_groups_0"), val = int32(1)]; + tensor e_1 = conv(dilations = e_1_dilations_0, groups = e_1_groups_0, pad = e_1_pad_0, pad_type = e_1_pad_type_0, strides = e_1_strides_0, weight = model_model_layers_0_mlp_down_proj_weight_palettized, x = input_17)[name = string("e_1")]; + tensor var_1281_axes_0 = const()[name = string("op_1281_axes_0"), val = tensor([2])]; + tensor var_1281 = squeeze(axes = var_1281_axes_0, x = e_1)[name = string("op_1281")]; + tensor var_1282 = const()[name = string("op_1282"), val = tensor([0, 2, 1])]; + tensor var_1283 = transpose(perm = var_1282, x = var_1281)[name = string("transpose_78")]; + tensor hidden_states_11_cast_fp16 = add(x = hidden_states_9_cast_fp16, y = var_1283)[name = string("hidden_states_11_cast_fp16")]; + int32 var_1295 = const()[name = string("op_1295"), val = int32(-1)]; + fp16 const_30_promoted_to_fp16 = const()[name = string("const_30_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1297_cast_fp16 = mul(x = hidden_states_11_cast_fp16, y = const_30_promoted_to_fp16)[name = string("op_1297_cast_fp16")]; + bool input_19_interleave_0 = const()[name = string("input_19_interleave_0"), val = bool(false)]; + tensor input_19_cast_fp16 = concat(axis = var_1295, interleave = input_19_interleave_0, values = (hidden_states_11_cast_fp16, var_1297_cast_fp16))[name = string("input_19_cast_fp16")]; + tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; + fp16 var_1292_to_fp16 = const()[name = string("op_1292_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_1292_to_fp16, x = input_19_cast_fp16)[name = string("normed_17_cast_fp16")]; + tensor normed_19_begin_0 = const()[name = string("normed_19_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_19_end_0 = const()[name = string("normed_19_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_19_end_mask_0 = const()[name = string("normed_19_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_19_cast_fp16 = slice_by_index(begin = normed_19_begin_0, end = normed_19_end_0, end_mask = normed_19_end_mask_0, x = normed_17_cast_fp16)[name = string("normed_19_cast_fp16")]; + tensor const_33_promoted_to_fp16 = const()[name = string("const_33_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667831872)))]; + tensor hidden_states_13_cast_fp16 = mul(x = normed_19_cast_fp16, y = const_33_promoted_to_fp16)[name = string("hidden_states_13_cast_fp16")]; + tensor var_1314 = const()[name = string("op_1314"), val = tensor([0, 2, 1])]; + tensor var_1317_axes_0 = const()[name = string("op_1317_axes_0"), val = tensor([2])]; + tensor var_1315_cast_fp16 = transpose(perm = var_1314, x = hidden_states_13_cast_fp16)[name = string("transpose_77")]; + tensor var_1317_cast_fp16 = expand_dims(axes = var_1317_axes_0, x = var_1315_cast_fp16)[name = string("op_1317_cast_fp16")]; + string var_1333_pad_type_0 = const()[name = string("op_1333_pad_type_0"), val = string("valid")]; + tensor var_1333_strides_0 = const()[name = string("op_1333_strides_0"), val = tensor([1, 1])]; + tensor var_1333_pad_0 = const()[name = string("op_1333_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1333_dilations_0 = const()[name = string("op_1333_dilations_0"), val = tensor([1, 1])]; + int32 var_1333_groups_0 = const()[name = string("op_1333_groups_0"), val = int32(1)]; + tensor var_1333 = conv(dilations = var_1333_dilations_0, groups = var_1333_groups_0, pad = var_1333_pad_0, pad_type = var_1333_pad_type_0, strides = var_1333_strides_0, weight = model_model_layers_1_self_attn_q_proj_weight_palettized, x = var_1317_cast_fp16)[name = string("op_1333")]; + tensor var_1338 = const()[name = string("op_1338"), val = tensor([1, 16, 1, 128])]; + tensor var_1339 = reshape(shape = var_1338, x = var_1333)[name = string("op_1339")]; + string var_1355_pad_type_0 = const()[name = string("op_1355_pad_type_0"), val = string("valid")]; + tensor var_1355_strides_0 = const()[name = string("op_1355_strides_0"), val = tensor([1, 1])]; + tensor var_1355_pad_0 = const()[name = string("op_1355_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1355_dilations_0 = const()[name = string("op_1355_dilations_0"), val = tensor([1, 1])]; + int32 var_1355_groups_0 = const()[name = string("op_1355_groups_0"), val = int32(1)]; + tensor var_1355 = conv(dilations = var_1355_dilations_0, groups = var_1355_groups_0, pad = var_1355_pad_0, pad_type = var_1355_pad_type_0, strides = var_1355_strides_0, weight = model_model_layers_1_self_attn_k_proj_weight_palettized, x = var_1317_cast_fp16)[name = string("op_1355")]; + tensor var_1360 = const()[name = string("op_1360"), val = tensor([1, 8, 1, 128])]; + tensor var_1361 = reshape(shape = var_1360, x = var_1355)[name = string("op_1361")]; + string var_1377_pad_type_0 = const()[name = string("op_1377_pad_type_0"), val = string("valid")]; + tensor var_1377_strides_0 = const()[name = string("op_1377_strides_0"), val = tensor([1, 1])]; + tensor var_1377_pad_0 = const()[name = string("op_1377_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1377_dilations_0 = const()[name = string("op_1377_dilations_0"), val = tensor([1, 1])]; + int32 var_1377_groups_0 = const()[name = string("op_1377_groups_0"), val = int32(1)]; + tensor var_1377 = conv(dilations = var_1377_dilations_0, groups = var_1377_groups_0, pad = var_1377_pad_0, pad_type = var_1377_pad_type_0, strides = var_1377_strides_0, weight = model_model_layers_1_self_attn_v_proj_weight_palettized, x = var_1317_cast_fp16)[name = string("op_1377")]; + tensor var_1382 = const()[name = string("op_1382"), val = tensor([1, 8, 1, 128])]; + tensor var_1383 = reshape(shape = var_1382, x = var_1377)[name = string("op_1383")]; + int32 var_1398 = const()[name = string("op_1398"), val = int32(-1)]; + fp16 const_34_promoted = const()[name = string("const_34_promoted"), val = fp16(-0x1p+0)]; + tensor var_1400 = mul(x = var_1339, y = const_34_promoted)[name = string("op_1400")]; + bool input_23_interleave_0 = const()[name = string("input_23_interleave_0"), val = bool(false)]; + tensor input_23 = concat(axis = var_1398, interleave = input_23_interleave_0, values = (var_1339, var_1400))[name = string("input_23")]; + tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; + fp16 var_1395_to_fp16 = const()[name = string("op_1395_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_1395_to_fp16, x = input_23)[name = string("normed_21_cast_fp16")]; + tensor normed_23_begin_0 = const()[name = string("normed_23_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_23_end_0 = const()[name = string("normed_23_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_23_end_mask_0 = const()[name = string("normed_23_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_23 = slice_by_index(begin = normed_23_begin_0, end = normed_23_end_0, end_mask = normed_23_end_mask_0, x = normed_21_cast_fp16)[name = string("normed_23")]; + tensor const_37 = const()[name = string("const_37"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667836032)))]; + tensor q_3 = mul(x = normed_23, y = const_37)[name = string("q_3")]; + int32 var_1423 = const()[name = string("op_1423"), val = int32(-1)]; + fp16 const_38_promoted = const()[name = string("const_38_promoted"), val = fp16(-0x1p+0)]; + tensor var_1425 = mul(x = var_1361, y = const_38_promoted)[name = string("op_1425")]; + bool input_25_interleave_0 = const()[name = string("input_25_interleave_0"), val = bool(false)]; + tensor input_25 = concat(axis = var_1423, interleave = input_25_interleave_0, values = (var_1361, var_1425))[name = string("input_25")]; + tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; + fp16 var_1420_to_fp16 = const()[name = string("op_1420_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_1420_to_fp16, x = input_25)[name = string("normed_25_cast_fp16")]; + tensor normed_27_begin_0 = const()[name = string("normed_27_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_27_end_0 = const()[name = string("normed_27_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_27_end_mask_0 = const()[name = string("normed_27_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_27 = slice_by_index(begin = normed_27_begin_0, end = normed_27_end_0, end_mask = normed_27_end_mask_0, x = normed_25_cast_fp16)[name = string("normed_27")]; + tensor const_41 = const()[name = string("const_41"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667836352)))]; + tensor k_3 = mul(x = normed_27, y = const_41)[name = string("k_3")]; + tensor var_1439 = mul(x = q_3, y = cos_1_cast_fp16)[name = string("op_1439")]; + tensor x1_5_begin_0 = const()[name = string("x1_5_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_5_end_0 = const()[name = string("x1_5_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_5_end_mask_0 = const()[name = string("x1_5_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_5 = slice_by_index(begin = x1_5_begin_0, end = x1_5_end_0, end_mask = x1_5_end_mask_0, x = q_3)[name = string("x1_5")]; + tensor x2_5_begin_0 = const()[name = string("x2_5_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_5_end_0 = const()[name = string("x2_5_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_5_end_mask_0 = const()[name = string("x2_5_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_5 = slice_by_index(begin = x2_5_begin_0, end = x2_5_end_0, end_mask = x2_5_end_mask_0, x = q_3)[name = string("x2_5")]; + fp16 const_44_promoted = const()[name = string("const_44_promoted"), val = fp16(-0x1p+0)]; + tensor var_1460 = mul(x = x2_5, y = const_44_promoted)[name = string("op_1460")]; + int32 var_1462 = const()[name = string("op_1462"), val = int32(-1)]; + bool var_1463_interleave_0 = const()[name = string("op_1463_interleave_0"), val = bool(false)]; + tensor var_1463 = concat(axis = var_1462, interleave = var_1463_interleave_0, values = (var_1460, x1_5))[name = string("op_1463")]; + tensor var_1464 = mul(x = var_1463, y = sin_1_cast_fp16)[name = string("op_1464")]; + tensor query_states_5 = add(x = var_1439, y = var_1464)[name = string("query_states_5")]; + tensor var_1467 = mul(x = k_3, y = cos_1_cast_fp16)[name = string("op_1467")]; + tensor x1_7_begin_0 = const()[name = string("x1_7_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_7_end_0 = const()[name = string("x1_7_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_7_end_mask_0 = const()[name = string("x1_7_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_7 = slice_by_index(begin = x1_7_begin_0, end = x1_7_end_0, end_mask = x1_7_end_mask_0, x = k_3)[name = string("x1_7")]; + tensor x2_7_begin_0 = const()[name = string("x2_7_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_7_end_0 = const()[name = string("x2_7_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_7_end_mask_0 = const()[name = string("x2_7_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_7 = slice_by_index(begin = x2_7_begin_0, end = x2_7_end_0, end_mask = x2_7_end_mask_0, x = k_3)[name = string("x2_7")]; + fp16 const_47_promoted = const()[name = string("const_47_promoted"), val = fp16(-0x1p+0)]; + tensor var_1488 = mul(x = x2_7, y = const_47_promoted)[name = string("op_1488")]; + int32 var_1490 = const()[name = string("op_1490"), val = int32(-1)]; + bool var_1491_interleave_0 = const()[name = string("op_1491_interleave_0"), val = bool(false)]; + tensor var_1491 = concat(axis = var_1490, interleave = var_1491_interleave_0, values = (var_1488, x1_7))[name = string("op_1491")]; + tensor var_1492 = mul(x = var_1491, y = sin_1_cast_fp16)[name = string("op_1492")]; + tensor key_states_5 = add(x = var_1467, y = var_1492)[name = string("key_states_5")]; + tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([1])]; + tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; + tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; + tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([2])]; + int32 concat_10_axis_0 = const()[name = string("concat_10_axis_0"), val = int32(0)]; + bool concat_10_interleave_0 = const()[name = string("concat_10_interleave_0"), val = bool(false)]; + tensor concat_10 = concat(axis = concat_10_axis_0, interleave = concat_10_interleave_0, values = (expand_dims_12, expand_dims_13, current_pos, expand_dims_15))[name = string("concat_10")]; + tensor concat_11_values1_0 = const()[name = string("concat_11_values1_0"), val = tensor([0])]; + tensor concat_11_values3_0 = const()[name = string("concat_11_values3_0"), val = tensor([0])]; + int32 concat_11_axis_0 = const()[name = string("concat_11_axis_0"), val = int32(0)]; + bool concat_11_interleave_0 = const()[name = string("concat_11_interleave_0"), val = bool(false)]; + tensor concat_11 = concat(axis = concat_11_axis_0, interleave = concat_11_interleave_0, values = (expand_dims_16, concat_11_values1_0, var_1001, concat_11_values3_0))[name = string("concat_11")]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_10, begin_mask = model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0, end = concat_11, end_mask = model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_3_stride_0, update = key_states_5, x = coreml_update_state_29)[name = string("model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_58_write_state")]; + tensor coreml_update_state_30 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_58")]; + tensor expand_dims_18 = const()[name = string("expand_dims_18"), val = tensor([29])]; + tensor expand_dims_19 = const()[name = string("expand_dims_19"), val = tensor([0])]; + tensor expand_dims_21 = const()[name = string("expand_dims_21"), val = tensor([0])]; + tensor expand_dims_22 = const()[name = string("expand_dims_22"), val = tensor([30])]; + int32 concat_14_axis_0 = const()[name = string("concat_14_axis_0"), val = int32(0)]; + bool concat_14_interleave_0 = const()[name = string("concat_14_interleave_0"), val = bool(false)]; + tensor concat_14 = concat(axis = concat_14_axis_0, interleave = concat_14_interleave_0, values = (expand_dims_18, expand_dims_19, current_pos, expand_dims_21))[name = string("concat_14")]; + tensor concat_15_values1_0 = const()[name = string("concat_15_values1_0"), val = tensor([0])]; + tensor concat_15_values3_0 = const()[name = string("concat_15_values3_0"), val = tensor([0])]; + int32 concat_15_axis_0 = const()[name = string("concat_15_axis_0"), val = int32(0)]; + bool concat_15_interleave_0 = const()[name = string("concat_15_interleave_0"), val = bool(false)]; + tensor concat_15 = concat(axis = concat_15_axis_0, interleave = concat_15_interleave_0, values = (expand_dims_22, concat_15_values1_0, var_1001, concat_15_values3_0))[name = string("concat_15")]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_14, begin_mask = model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0, end = concat_15, end_mask = model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_4_stride_0, update = var_1383, x = coreml_update_state_30)[name = string("model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_59_write_state")]; + tensor coreml_update_state_31 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_59")]; + tensor var_1547_begin_0 = const()[name = string("op_1547_begin_0"), val = tensor([1, 0, 0, 0])]; + tensor var_1547_end_0 = const()[name = string("op_1547_end_0"), val = tensor([2, 8, 1024, 128])]; + tensor var_1547_end_mask_0 = const()[name = string("op_1547_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1547_cast_fp16 = slice_by_index(begin = var_1547_begin_0, end = var_1547_end_0, end_mask = var_1547_end_mask_0, x = coreml_update_state_31)[name = string("op_1547_cast_fp16")]; + tensor K_layer_cache_3_axes_0 = const()[name = string("K_layer_cache_3_axes_0"), val = tensor([0])]; + tensor K_layer_cache_3_cast_fp16 = squeeze(axes = K_layer_cache_3_axes_0, x = var_1547_cast_fp16)[name = string("K_layer_cache_3_cast_fp16")]; + tensor var_1554_begin_0 = const()[name = string("op_1554_begin_0"), val = tensor([29, 0, 0, 0])]; + tensor var_1554_end_0 = const()[name = string("op_1554_end_0"), val = tensor([30, 8, 1024, 128])]; + tensor var_1554_end_mask_0 = const()[name = string("op_1554_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1554_cast_fp16 = slice_by_index(begin = var_1554_begin_0, end = var_1554_end_0, end_mask = var_1554_end_mask_0, x = coreml_update_state_31)[name = string("op_1554_cast_fp16")]; + tensor V_layer_cache_3_axes_0 = const()[name = string("V_layer_cache_3_axes_0"), val = tensor([0])]; + tensor V_layer_cache_3_cast_fp16 = squeeze(axes = V_layer_cache_3_axes_0, x = var_1554_cast_fp16)[name = string("V_layer_cache_3_cast_fp16")]; + tensor x_19_axes_0 = const()[name = string("x_19_axes_0"), val = tensor([1])]; + tensor x_19_cast_fp16 = expand_dims(axes = x_19_axes_0, x = K_layer_cache_3_cast_fp16)[name = string("x_19_cast_fp16")]; + tensor var_1591 = const()[name = string("op_1591"), val = tensor([1, 2, 1, 1])]; + tensor x_21_cast_fp16 = tile(reps = var_1591, x = x_19_cast_fp16)[name = string("x_21_cast_fp16")]; + tensor var_1603 = const()[name = string("op_1603"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_7_cast_fp16 = reshape(shape = var_1603, x = x_21_cast_fp16)[name = string("key_states_7_cast_fp16")]; + tensor x_25_axes_0 = const()[name = string("x_25_axes_0"), val = tensor([1])]; + tensor x_25_cast_fp16 = expand_dims(axes = x_25_axes_0, x = V_layer_cache_3_cast_fp16)[name = string("x_25_cast_fp16")]; + tensor var_1611 = const()[name = string("op_1611"), val = tensor([1, 2, 1, 1])]; + tensor x_27_cast_fp16 = tile(reps = var_1611, x = x_25_cast_fp16)[name = string("x_27_cast_fp16")]; + tensor var_1623 = const()[name = string("op_1623"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_9_cast_fp16 = reshape(shape = var_1623, x = x_27_cast_fp16)[name = string("value_states_9_cast_fp16")]; + bool var_1638_transpose_x_1 = const()[name = string("op_1638_transpose_x_1"), val = bool(false)]; + bool var_1638_transpose_y_1 = const()[name = string("op_1638_transpose_y_1"), val = bool(true)]; + tensor var_1638 = matmul(transpose_x = var_1638_transpose_x_1, transpose_y = var_1638_transpose_y_1, x = query_states_5, y = key_states_7_cast_fp16)[name = string("op_1638")]; + fp16 var_1639_to_fp16 = const()[name = string("op_1639_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_7_cast_fp16 = mul(x = var_1638, y = var_1639_to_fp16)[name = string("attn_weights_7_cast_fp16")]; + tensor attn_weights_9_cast_fp16 = add(x = attn_weights_7_cast_fp16, y = causal_mask)[name = string("attn_weights_9_cast_fp16")]; + int32 var_1674 = const()[name = string("op_1674"), val = int32(-1)]; + tensor attn_weights_11_cast_fp16 = softmax(axis = var_1674, x = attn_weights_9_cast_fp16)[name = string("attn_weights_11_cast_fp16")]; + bool attn_output_11_transpose_x_0 = const()[name = string("attn_output_11_transpose_x_0"), val = bool(false)]; + bool attn_output_11_transpose_y_0 = const()[name = string("attn_output_11_transpose_y_0"), val = bool(false)]; + tensor attn_output_11_cast_fp16 = matmul(transpose_x = attn_output_11_transpose_x_0, transpose_y = attn_output_11_transpose_y_0, x = attn_weights_11_cast_fp16, y = value_states_9_cast_fp16)[name = string("attn_output_11_cast_fp16")]; + tensor var_1685_perm_0 = const()[name = string("op_1685_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1689 = const()[name = string("op_1689"), val = tensor([1, 1, 2048])]; + tensor var_1685_cast_fp16 = transpose(perm = var_1685_perm_0, x = attn_output_11_cast_fp16)[name = string("transpose_76")]; + tensor attn_output_15_cast_fp16 = reshape(shape = var_1689, x = var_1685_cast_fp16)[name = string("attn_output_15_cast_fp16")]; + tensor var_1694 = const()[name = string("op_1694"), val = tensor([0, 2, 1])]; + string var_1710_pad_type_0 = const()[name = string("op_1710_pad_type_0"), val = string("valid")]; + int32 var_1710_groups_0 = const()[name = string("op_1710_groups_0"), val = int32(1)]; + tensor var_1710_strides_0 = const()[name = string("op_1710_strides_0"), val = tensor([1])]; + tensor var_1710_pad_0 = const()[name = string("op_1710_pad_0"), val = tensor([0, 0])]; + tensor var_1710_dilations_0 = const()[name = string("op_1710_dilations_0"), val = tensor([1])]; + tensor squeeze_1_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667836672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672031040))))[name = string("squeeze_1_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_1695_cast_fp16 = transpose(perm = var_1694, x = attn_output_15_cast_fp16)[name = string("transpose_75")]; + tensor var_1710_cast_fp16 = conv(dilations = var_1710_dilations_0, groups = var_1710_groups_0, pad = var_1710_pad_0, pad_type = var_1710_pad_type_0, strides = var_1710_strides_0, weight = squeeze_1_cast_fp16_to_fp32_to_fp16_palettized, x = var_1695_cast_fp16)[name = string("op_1710_cast_fp16")]; + tensor var_1714 = const()[name = string("op_1714"), val = tensor([0, 2, 1])]; + tensor attn_output_19_cast_fp16 = transpose(perm = var_1714, x = var_1710_cast_fp16)[name = string("transpose_74")]; + tensor hidden_states_19_cast_fp16 = add(x = hidden_states_11_cast_fp16, y = attn_output_19_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; + int32 var_1727 = const()[name = string("op_1727"), val = int32(-1)]; + fp16 const_56_promoted_to_fp16 = const()[name = string("const_56_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1729_cast_fp16 = mul(x = hidden_states_19_cast_fp16, y = const_56_promoted_to_fp16)[name = string("op_1729_cast_fp16")]; + bool input_29_interleave_0 = const()[name = string("input_29_interleave_0"), val = bool(false)]; + tensor input_29_cast_fp16 = concat(axis = var_1727, interleave = input_29_interleave_0, values = (hidden_states_19_cast_fp16, var_1729_cast_fp16))[name = string("input_29_cast_fp16")]; + tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; + fp16 var_1724_to_fp16 = const()[name = string("op_1724_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_1724_to_fp16, x = input_29_cast_fp16)[name = string("normed_29_cast_fp16")]; + tensor normed_31_begin_0 = const()[name = string("normed_31_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_31_end_0 = const()[name = string("normed_31_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_31_end_mask_0 = const()[name = string("normed_31_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_31_cast_fp16 = slice_by_index(begin = normed_31_begin_0, end = normed_31_end_0, end_mask = normed_31_end_mask_0, x = normed_29_cast_fp16)[name = string("normed_31_cast_fp16")]; + tensor const_59_promoted_to_fp16 = const()[name = string("const_59_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672162176)))]; + tensor x_29_cast_fp16 = mul(x = normed_31_cast_fp16, y = const_59_promoted_to_fp16)[name = string("x_29_cast_fp16")]; + tensor var_1754 = const()[name = string("op_1754"), val = tensor([0, 2, 1])]; + tensor input_31_axes_0 = const()[name = string("input_31_axes_0"), val = tensor([2])]; + tensor var_1755 = transpose(perm = var_1754, x = x_29_cast_fp16)[name = string("transpose_73")]; + tensor input_31 = expand_dims(axes = input_31_axes_0, x = var_1755)[name = string("input_31")]; + string input_33_pad_type_0 = const()[name = string("input_33_pad_type_0"), val = string("valid")]; + tensor input_33_strides_0 = const()[name = string("input_33_strides_0"), val = tensor([1, 1])]; + tensor input_33_pad_0 = const()[name = string("input_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_33_dilations_0 = const()[name = string("input_33_dilations_0"), val = tensor([1, 1])]; + int32 input_33_groups_0 = const()[name = string("input_33_groups_0"), val = int32(1)]; + tensor input_33 = conv(dilations = input_33_dilations_0, groups = input_33_groups_0, pad = input_33_pad_0, pad_type = input_33_pad_type_0, strides = input_33_strides_0, weight = model_model_layers_1_mlp_gate_proj_weight_palettized, x = input_31)[name = string("input_33")]; + string b_3_pad_type_0 = const()[name = string("b_3_pad_type_0"), val = string("valid")]; + tensor b_3_strides_0 = const()[name = string("b_3_strides_0"), val = tensor([1, 1])]; + tensor b_3_pad_0 = const()[name = string("b_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_3_dilations_0 = const()[name = string("b_3_dilations_0"), val = tensor([1, 1])]; + int32 b_3_groups_0 = const()[name = string("b_3_groups_0"), val = int32(1)]; + tensor b_3 = conv(dilations = b_3_dilations_0, groups = b_3_groups_0, pad = b_3_pad_0, pad_type = b_3_pad_type_0, strides = b_3_strides_0, weight = model_model_layers_1_mlp_up_proj_weight_palettized, x = input_31)[name = string("b_3")]; + tensor c_3 = silu(x = input_33)[name = string("c_3")]; + tensor input_35 = mul(x = c_3, y = b_3)[name = string("input_35")]; + string e_3_pad_type_0 = const()[name = string("e_3_pad_type_0"), val = string("valid")]; + tensor e_3_strides_0 = const()[name = string("e_3_strides_0"), val = tensor([1, 1])]; + tensor e_3_pad_0 = const()[name = string("e_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_3_dilations_0 = const()[name = string("e_3_dilations_0"), val = tensor([1, 1])]; + int32 e_3_groups_0 = const()[name = string("e_3_groups_0"), val = int32(1)]; + tensor e_3 = conv(dilations = e_3_dilations_0, groups = e_3_groups_0, pad = e_3_pad_0, pad_type = e_3_pad_type_0, strides = e_3_strides_0, weight = model_model_layers_1_mlp_down_proj_weight_palettized, x = input_35)[name = string("e_3")]; + tensor var_1777_axes_0 = const()[name = string("op_1777_axes_0"), val = tensor([2])]; + tensor var_1777 = squeeze(axes = var_1777_axes_0, x = e_3)[name = string("op_1777")]; + tensor var_1778 = const()[name = string("op_1778"), val = tensor([0, 2, 1])]; + tensor var_1779 = transpose(perm = var_1778, x = var_1777)[name = string("transpose_72")]; + tensor hidden_states_21_cast_fp16 = add(x = hidden_states_19_cast_fp16, y = var_1779)[name = string("hidden_states_21_cast_fp16")]; + int32 var_1791 = const()[name = string("op_1791"), val = int32(-1)]; + fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1793_cast_fp16 = mul(x = hidden_states_21_cast_fp16, y = const_60_promoted_to_fp16)[name = string("op_1793_cast_fp16")]; + bool input_37_interleave_0 = const()[name = string("input_37_interleave_0"), val = bool(false)]; + tensor input_37_cast_fp16 = concat(axis = var_1791, interleave = input_37_interleave_0, values = (hidden_states_21_cast_fp16, var_1793_cast_fp16))[name = string("input_37_cast_fp16")]; + tensor normed_33_axes_0 = const()[name = string("normed_33_axes_0"), val = tensor([-1])]; + fp16 var_1788_to_fp16 = const()[name = string("op_1788_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_33_cast_fp16 = layer_norm(axes = normed_33_axes_0, epsilon = var_1788_to_fp16, x = input_37_cast_fp16)[name = string("normed_33_cast_fp16")]; + tensor normed_35_begin_0 = const()[name = string("normed_35_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_35_end_0 = const()[name = string("normed_35_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_35_end_mask_0 = const()[name = string("normed_35_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_35_cast_fp16 = slice_by_index(begin = normed_35_begin_0, end = normed_35_end_0, end_mask = normed_35_end_mask_0, x = normed_33_cast_fp16)[name = string("normed_35_cast_fp16")]; + tensor const_63_promoted_to_fp16 = const()[name = string("const_63_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672166336)))]; + tensor hidden_states_23_cast_fp16 = mul(x = normed_35_cast_fp16, y = const_63_promoted_to_fp16)[name = string("hidden_states_23_cast_fp16")]; + tensor var_1810 = const()[name = string("op_1810"), val = tensor([0, 2, 1])]; + tensor var_1813_axes_0 = const()[name = string("op_1813_axes_0"), val = tensor([2])]; + tensor var_1811_cast_fp16 = transpose(perm = var_1810, x = hidden_states_23_cast_fp16)[name = string("transpose_71")]; + tensor var_1813_cast_fp16 = expand_dims(axes = var_1813_axes_0, x = var_1811_cast_fp16)[name = string("op_1813_cast_fp16")]; + string var_1829_pad_type_0 = const()[name = string("op_1829_pad_type_0"), val = string("valid")]; + tensor var_1829_strides_0 = const()[name = string("op_1829_strides_0"), val = tensor([1, 1])]; + tensor var_1829_pad_0 = const()[name = string("op_1829_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1829_dilations_0 = const()[name = string("op_1829_dilations_0"), val = tensor([1, 1])]; + int32 var_1829_groups_0 = const()[name = string("op_1829_groups_0"), val = int32(1)]; + tensor var_1829 = conv(dilations = var_1829_dilations_0, groups = var_1829_groups_0, pad = var_1829_pad_0, pad_type = var_1829_pad_type_0, strides = var_1829_strides_0, weight = model_model_layers_2_self_attn_q_proj_weight_palettized, x = var_1813_cast_fp16)[name = string("op_1829")]; + tensor var_1834 = const()[name = string("op_1834"), val = tensor([1, 16, 1, 128])]; + tensor var_1835 = reshape(shape = var_1834, x = var_1829)[name = string("op_1835")]; + string var_1851_pad_type_0 = const()[name = string("op_1851_pad_type_0"), val = string("valid")]; + tensor var_1851_strides_0 = const()[name = string("op_1851_strides_0"), val = tensor([1, 1])]; + tensor var_1851_pad_0 = const()[name = string("op_1851_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1851_dilations_0 = const()[name = string("op_1851_dilations_0"), val = tensor([1, 1])]; + int32 var_1851_groups_0 = const()[name = string("op_1851_groups_0"), val = int32(1)]; + tensor var_1851 = conv(dilations = var_1851_dilations_0, groups = var_1851_groups_0, pad = var_1851_pad_0, pad_type = var_1851_pad_type_0, strides = var_1851_strides_0, weight = model_model_layers_2_self_attn_k_proj_weight_palettized, x = var_1813_cast_fp16)[name = string("op_1851")]; + tensor var_1856 = const()[name = string("op_1856"), val = tensor([1, 8, 1, 128])]; + tensor var_1857 = reshape(shape = var_1856, x = var_1851)[name = string("op_1857")]; + string var_1873_pad_type_0 = const()[name = string("op_1873_pad_type_0"), val = string("valid")]; + tensor var_1873_strides_0 = const()[name = string("op_1873_strides_0"), val = tensor([1, 1])]; + tensor var_1873_pad_0 = const()[name = string("op_1873_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1873_dilations_0 = const()[name = string("op_1873_dilations_0"), val = tensor([1, 1])]; + int32 var_1873_groups_0 = const()[name = string("op_1873_groups_0"), val = int32(1)]; + tensor var_1873 = conv(dilations = var_1873_dilations_0, groups = var_1873_groups_0, pad = var_1873_pad_0, pad_type = var_1873_pad_type_0, strides = var_1873_strides_0, weight = model_model_layers_2_self_attn_v_proj_weight_palettized, x = var_1813_cast_fp16)[name = string("op_1873")]; + tensor var_1878 = const()[name = string("op_1878"), val = tensor([1, 8, 1, 128])]; + tensor var_1879 = reshape(shape = var_1878, x = var_1873)[name = string("op_1879")]; + int32 var_1894 = const()[name = string("op_1894"), val = int32(-1)]; + fp16 const_64_promoted = const()[name = string("const_64_promoted"), val = fp16(-0x1p+0)]; + tensor var_1896 = mul(x = var_1835, y = const_64_promoted)[name = string("op_1896")]; + bool input_41_interleave_0 = const()[name = string("input_41_interleave_0"), val = bool(false)]; + tensor input_41 = concat(axis = var_1894, interleave = input_41_interleave_0, values = (var_1835, var_1896))[name = string("input_41")]; + tensor normed_37_axes_0 = const()[name = string("normed_37_axes_0"), val = tensor([-1])]; + fp16 var_1891_to_fp16 = const()[name = string("op_1891_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_37_cast_fp16 = layer_norm(axes = normed_37_axes_0, epsilon = var_1891_to_fp16, x = input_41)[name = string("normed_37_cast_fp16")]; + tensor normed_39_begin_0 = const()[name = string("normed_39_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_39_end_0 = const()[name = string("normed_39_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_39_end_mask_0 = const()[name = string("normed_39_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_39 = slice_by_index(begin = normed_39_begin_0, end = normed_39_end_0, end_mask = normed_39_end_mask_0, x = normed_37_cast_fp16)[name = string("normed_39")]; + tensor const_67 = const()[name = string("const_67"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672170496)))]; + tensor q_5 = mul(x = normed_39, y = const_67)[name = string("q_5")]; + int32 var_1919 = const()[name = string("op_1919"), val = int32(-1)]; + fp16 const_68_promoted = const()[name = string("const_68_promoted"), val = fp16(-0x1p+0)]; + tensor var_1921 = mul(x = var_1857, y = const_68_promoted)[name = string("op_1921")]; + bool input_43_interleave_0 = const()[name = string("input_43_interleave_0"), val = bool(false)]; + tensor input_43 = concat(axis = var_1919, interleave = input_43_interleave_0, values = (var_1857, var_1921))[name = string("input_43")]; + tensor normed_41_axes_0 = const()[name = string("normed_41_axes_0"), val = tensor([-1])]; + fp16 var_1916_to_fp16 = const()[name = string("op_1916_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_41_cast_fp16 = layer_norm(axes = normed_41_axes_0, epsilon = var_1916_to_fp16, x = input_43)[name = string("normed_41_cast_fp16")]; + tensor normed_43_begin_0 = const()[name = string("normed_43_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_43_end_0 = const()[name = string("normed_43_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_43_end_mask_0 = const()[name = string("normed_43_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_43 = slice_by_index(begin = normed_43_begin_0, end = normed_43_end_0, end_mask = normed_43_end_mask_0, x = normed_41_cast_fp16)[name = string("normed_43")]; + tensor const_71 = const()[name = string("const_71"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672170816)))]; + tensor k_5 = mul(x = normed_43, y = const_71)[name = string("k_5")]; + tensor var_1935 = mul(x = q_5, y = cos_1_cast_fp16)[name = string("op_1935")]; + tensor x1_9_begin_0 = const()[name = string("x1_9_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_9_end_0 = const()[name = string("x1_9_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_9_end_mask_0 = const()[name = string("x1_9_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_9 = slice_by_index(begin = x1_9_begin_0, end = x1_9_end_0, end_mask = x1_9_end_mask_0, x = q_5)[name = string("x1_9")]; + tensor x2_9_begin_0 = const()[name = string("x2_9_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_9_end_0 = const()[name = string("x2_9_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_9_end_mask_0 = const()[name = string("x2_9_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_9 = slice_by_index(begin = x2_9_begin_0, end = x2_9_end_0, end_mask = x2_9_end_mask_0, x = q_5)[name = string("x2_9")]; + fp16 const_74_promoted = const()[name = string("const_74_promoted"), val = fp16(-0x1p+0)]; + tensor var_1956 = mul(x = x2_9, y = const_74_promoted)[name = string("op_1956")]; + int32 var_1958 = const()[name = string("op_1958"), val = int32(-1)]; + bool var_1959_interleave_0 = const()[name = string("op_1959_interleave_0"), val = bool(false)]; + tensor var_1959 = concat(axis = var_1958, interleave = var_1959_interleave_0, values = (var_1956, x1_9))[name = string("op_1959")]; + tensor var_1960 = mul(x = var_1959, y = sin_1_cast_fp16)[name = string("op_1960")]; + tensor query_states_9 = add(x = var_1935, y = var_1960)[name = string("query_states_9")]; + tensor var_1963 = mul(x = k_5, y = cos_1_cast_fp16)[name = string("op_1963")]; + tensor x1_11_begin_0 = const()[name = string("x1_11_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_11_end_0 = const()[name = string("x1_11_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_11_end_mask_0 = const()[name = string("x1_11_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_11 = slice_by_index(begin = x1_11_begin_0, end = x1_11_end_0, end_mask = x1_11_end_mask_0, x = k_5)[name = string("x1_11")]; + tensor x2_11_begin_0 = const()[name = string("x2_11_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_11_end_0 = const()[name = string("x2_11_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_11_end_mask_0 = const()[name = string("x2_11_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_11 = slice_by_index(begin = x2_11_begin_0, end = x2_11_end_0, end_mask = x2_11_end_mask_0, x = k_5)[name = string("x2_11")]; + fp16 const_77_promoted = const()[name = string("const_77_promoted"), val = fp16(-0x1p+0)]; + tensor var_1984 = mul(x = x2_11, y = const_77_promoted)[name = string("op_1984")]; + int32 var_1986 = const()[name = string("op_1986"), val = int32(-1)]; + bool var_1987_interleave_0 = const()[name = string("op_1987_interleave_0"), val = bool(false)]; + tensor var_1987 = concat(axis = var_1986, interleave = var_1987_interleave_0, values = (var_1984, x1_11))[name = string("op_1987")]; + tensor var_1988 = mul(x = var_1987, y = sin_1_cast_fp16)[name = string("op_1988")]; + tensor key_states_9 = add(x = var_1963, y = var_1988)[name = string("key_states_9")]; + tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([2])]; + tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; + tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; + tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([3])]; + int32 concat_18_axis_0 = const()[name = string("concat_18_axis_0"), val = int32(0)]; + bool concat_18_interleave_0 = const()[name = string("concat_18_interleave_0"), val = bool(false)]; + tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (expand_dims_24, expand_dims_25, current_pos, expand_dims_27))[name = string("concat_18")]; + tensor concat_19_values1_0 = const()[name = string("concat_19_values1_0"), val = tensor([0])]; + tensor concat_19_values3_0 = const()[name = string("concat_19_values3_0"), val = tensor([0])]; + int32 concat_19_axis_0 = const()[name = string("concat_19_axis_0"), val = int32(0)]; + bool concat_19_interleave_0 = const()[name = string("concat_19_interleave_0"), val = bool(false)]; + tensor concat_19 = concat(axis = concat_19_axis_0, interleave = concat_19_interleave_0, values = (expand_dims_28, concat_19_values1_0, var_1001, concat_19_values3_0))[name = string("concat_19")]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_18, begin_mask = model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0, end = concat_19, end_mask = model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_5_stride_0, update = key_states_9, x = coreml_update_state_31)[name = string("model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_60_write_state")]; + tensor coreml_update_state_32 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_60")]; + tensor expand_dims_30 = const()[name = string("expand_dims_30"), val = tensor([30])]; + tensor expand_dims_31 = const()[name = string("expand_dims_31"), val = tensor([0])]; + tensor expand_dims_33 = const()[name = string("expand_dims_33"), val = tensor([0])]; + tensor expand_dims_34 = const()[name = string("expand_dims_34"), val = tensor([31])]; + int32 concat_22_axis_0 = const()[name = string("concat_22_axis_0"), val = int32(0)]; + bool concat_22_interleave_0 = const()[name = string("concat_22_interleave_0"), val = bool(false)]; + tensor concat_22 = concat(axis = concat_22_axis_0, interleave = concat_22_interleave_0, values = (expand_dims_30, expand_dims_31, current_pos, expand_dims_33))[name = string("concat_22")]; + tensor concat_23_values1_0 = const()[name = string("concat_23_values1_0"), val = tensor([0])]; + tensor concat_23_values3_0 = const()[name = string("concat_23_values3_0"), val = tensor([0])]; + int32 concat_23_axis_0 = const()[name = string("concat_23_axis_0"), val = int32(0)]; + bool concat_23_interleave_0 = const()[name = string("concat_23_interleave_0"), val = bool(false)]; + tensor concat_23 = concat(axis = concat_23_axis_0, interleave = concat_23_interleave_0, values = (expand_dims_34, concat_23_values1_0, var_1001, concat_23_values3_0))[name = string("concat_23")]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_22, begin_mask = model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0, end = concat_23, end_mask = model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_6_stride_0, update = var_1879, x = coreml_update_state_32)[name = string("model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_61_write_state")]; + tensor coreml_update_state_33 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_61")]; + tensor var_2043_begin_0 = const()[name = string("op_2043_begin_0"), val = tensor([2, 0, 0, 0])]; + tensor var_2043_end_0 = const()[name = string("op_2043_end_0"), val = tensor([3, 8, 1024, 128])]; + tensor var_2043_end_mask_0 = const()[name = string("op_2043_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2043_cast_fp16 = slice_by_index(begin = var_2043_begin_0, end = var_2043_end_0, end_mask = var_2043_end_mask_0, x = coreml_update_state_33)[name = string("op_2043_cast_fp16")]; + tensor K_layer_cache_5_axes_0 = const()[name = string("K_layer_cache_5_axes_0"), val = tensor([0])]; + tensor K_layer_cache_5_cast_fp16 = squeeze(axes = K_layer_cache_5_axes_0, x = var_2043_cast_fp16)[name = string("K_layer_cache_5_cast_fp16")]; + tensor var_2050_begin_0 = const()[name = string("op_2050_begin_0"), val = tensor([30, 0, 0, 0])]; + tensor var_2050_end_0 = const()[name = string("op_2050_end_0"), val = tensor([31, 8, 1024, 128])]; + tensor var_2050_end_mask_0 = const()[name = string("op_2050_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2050_cast_fp16 = slice_by_index(begin = var_2050_begin_0, end = var_2050_end_0, end_mask = var_2050_end_mask_0, x = coreml_update_state_33)[name = string("op_2050_cast_fp16")]; + tensor V_layer_cache_5_axes_0 = const()[name = string("V_layer_cache_5_axes_0"), val = tensor([0])]; + tensor V_layer_cache_5_cast_fp16 = squeeze(axes = V_layer_cache_5_axes_0, x = var_2050_cast_fp16)[name = string("V_layer_cache_5_cast_fp16")]; + tensor x_35_axes_0 = const()[name = string("x_35_axes_0"), val = tensor([1])]; + tensor x_35_cast_fp16 = expand_dims(axes = x_35_axes_0, x = K_layer_cache_5_cast_fp16)[name = string("x_35_cast_fp16")]; + tensor var_2087 = const()[name = string("op_2087"), val = tensor([1, 2, 1, 1])]; + tensor x_37_cast_fp16 = tile(reps = var_2087, x = x_35_cast_fp16)[name = string("x_37_cast_fp16")]; + tensor var_2099 = const()[name = string("op_2099"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_11_cast_fp16 = reshape(shape = var_2099, x = x_37_cast_fp16)[name = string("key_states_11_cast_fp16")]; + tensor x_41_axes_0 = const()[name = string("x_41_axes_0"), val = tensor([1])]; + tensor x_41_cast_fp16 = expand_dims(axes = x_41_axes_0, x = V_layer_cache_5_cast_fp16)[name = string("x_41_cast_fp16")]; + tensor var_2107 = const()[name = string("op_2107"), val = tensor([1, 2, 1, 1])]; + tensor x_43_cast_fp16 = tile(reps = var_2107, x = x_41_cast_fp16)[name = string("x_43_cast_fp16")]; + tensor var_2119 = const()[name = string("op_2119"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_15_cast_fp16 = reshape(shape = var_2119, x = x_43_cast_fp16)[name = string("value_states_15_cast_fp16")]; + bool var_2134_transpose_x_1 = const()[name = string("op_2134_transpose_x_1"), val = bool(false)]; + bool var_2134_transpose_y_1 = const()[name = string("op_2134_transpose_y_1"), val = bool(true)]; + tensor var_2134 = matmul(transpose_x = var_2134_transpose_x_1, transpose_y = var_2134_transpose_y_1, x = query_states_9, y = key_states_11_cast_fp16)[name = string("op_2134")]; + fp16 var_2135_to_fp16 = const()[name = string("op_2135_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_13_cast_fp16 = mul(x = var_2134, y = var_2135_to_fp16)[name = string("attn_weights_13_cast_fp16")]; + tensor attn_weights_15_cast_fp16 = add(x = attn_weights_13_cast_fp16, y = causal_mask)[name = string("attn_weights_15_cast_fp16")]; + int32 var_2170 = const()[name = string("op_2170"), val = int32(-1)]; + tensor attn_weights_17_cast_fp16 = softmax(axis = var_2170, x = attn_weights_15_cast_fp16)[name = string("attn_weights_17_cast_fp16")]; + bool attn_output_21_transpose_x_0 = const()[name = string("attn_output_21_transpose_x_0"), val = bool(false)]; + bool attn_output_21_transpose_y_0 = const()[name = string("attn_output_21_transpose_y_0"), val = bool(false)]; + tensor attn_output_21_cast_fp16 = matmul(transpose_x = attn_output_21_transpose_x_0, transpose_y = attn_output_21_transpose_y_0, x = attn_weights_17_cast_fp16, y = value_states_15_cast_fp16)[name = string("attn_output_21_cast_fp16")]; + tensor var_2181_perm_0 = const()[name = string("op_2181_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_2185 = const()[name = string("op_2185"), val = tensor([1, 1, 2048])]; + tensor var_2181_cast_fp16 = transpose(perm = var_2181_perm_0, x = attn_output_21_cast_fp16)[name = string("transpose_70")]; + tensor attn_output_25_cast_fp16 = reshape(shape = var_2185, x = var_2181_cast_fp16)[name = string("attn_output_25_cast_fp16")]; + tensor var_2190 = const()[name = string("op_2190"), val = tensor([0, 2, 1])]; + string var_2206_pad_type_0 = const()[name = string("op_2206_pad_type_0"), val = string("valid")]; + int32 var_2206_groups_0 = const()[name = string("op_2206_groups_0"), val = int32(1)]; + tensor var_2206_strides_0 = const()[name = string("op_2206_strides_0"), val = tensor([1])]; + tensor var_2206_pad_0 = const()[name = string("op_2206_pad_0"), val = tensor([0, 0])]; + tensor var_2206_dilations_0 = const()[name = string("op_2206_dilations_0"), val = tensor([1])]; + tensor squeeze_2_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672171136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676365504))))[name = string("squeeze_2_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_2191_cast_fp16 = transpose(perm = var_2190, x = attn_output_25_cast_fp16)[name = string("transpose_69")]; + tensor var_2206_cast_fp16 = conv(dilations = var_2206_dilations_0, groups = var_2206_groups_0, pad = var_2206_pad_0, pad_type = var_2206_pad_type_0, strides = var_2206_strides_0, weight = squeeze_2_cast_fp16_to_fp32_to_fp16_palettized, x = var_2191_cast_fp16)[name = string("op_2206_cast_fp16")]; + tensor var_2210 = const()[name = string("op_2210"), val = tensor([0, 2, 1])]; + tensor attn_output_29_cast_fp16 = transpose(perm = var_2210, x = var_2206_cast_fp16)[name = string("transpose_68")]; + tensor hidden_states_29_cast_fp16 = add(x = hidden_states_21_cast_fp16, y = attn_output_29_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; + int32 var_2223 = const()[name = string("op_2223"), val = int32(-1)]; + fp16 const_86_promoted_to_fp16 = const()[name = string("const_86_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2225_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_86_promoted_to_fp16)[name = string("op_2225_cast_fp16")]; + bool input_47_interleave_0 = const()[name = string("input_47_interleave_0"), val = bool(false)]; + tensor input_47_cast_fp16 = concat(axis = var_2223, interleave = input_47_interleave_0, values = (hidden_states_29_cast_fp16, var_2225_cast_fp16))[name = string("input_47_cast_fp16")]; + tensor normed_45_axes_0 = const()[name = string("normed_45_axes_0"), val = tensor([-1])]; + fp16 var_2220_to_fp16 = const()[name = string("op_2220_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_45_cast_fp16 = layer_norm(axes = normed_45_axes_0, epsilon = var_2220_to_fp16, x = input_47_cast_fp16)[name = string("normed_45_cast_fp16")]; + tensor normed_47_begin_0 = const()[name = string("normed_47_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_47_end_0 = const()[name = string("normed_47_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_47_end_mask_0 = const()[name = string("normed_47_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_47_cast_fp16 = slice_by_index(begin = normed_47_begin_0, end = normed_47_end_0, end_mask = normed_47_end_mask_0, x = normed_45_cast_fp16)[name = string("normed_47_cast_fp16")]; + tensor const_89_promoted_to_fp16 = const()[name = string("const_89_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676496640)))]; + tensor x_45_cast_fp16 = mul(x = normed_47_cast_fp16, y = const_89_promoted_to_fp16)[name = string("x_45_cast_fp16")]; + tensor var_2250 = const()[name = string("op_2250"), val = tensor([0, 2, 1])]; + tensor input_49_axes_0 = const()[name = string("input_49_axes_0"), val = tensor([2])]; + tensor var_2251 = transpose(perm = var_2250, x = x_45_cast_fp16)[name = string("transpose_67")]; + tensor input_49 = expand_dims(axes = input_49_axes_0, x = var_2251)[name = string("input_49")]; + string input_51_pad_type_0 = const()[name = string("input_51_pad_type_0"), val = string("valid")]; + tensor input_51_strides_0 = const()[name = string("input_51_strides_0"), val = tensor([1, 1])]; + tensor input_51_pad_0 = const()[name = string("input_51_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_51_dilations_0 = const()[name = string("input_51_dilations_0"), val = tensor([1, 1])]; + int32 input_51_groups_0 = const()[name = string("input_51_groups_0"), val = int32(1)]; + tensor input_51 = conv(dilations = input_51_dilations_0, groups = input_51_groups_0, pad = input_51_pad_0, pad_type = input_51_pad_type_0, strides = input_51_strides_0, weight = model_model_layers_2_mlp_gate_proj_weight_palettized, x = input_49)[name = string("input_51")]; + string b_5_pad_type_0 = const()[name = string("b_5_pad_type_0"), val = string("valid")]; + tensor b_5_strides_0 = const()[name = string("b_5_strides_0"), val = tensor([1, 1])]; + tensor b_5_pad_0 = const()[name = string("b_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_5_dilations_0 = const()[name = string("b_5_dilations_0"), val = tensor([1, 1])]; + int32 b_5_groups_0 = const()[name = string("b_5_groups_0"), val = int32(1)]; + tensor b_5 = conv(dilations = b_5_dilations_0, groups = b_5_groups_0, pad = b_5_pad_0, pad_type = b_5_pad_type_0, strides = b_5_strides_0, weight = model_model_layers_2_mlp_up_proj_weight_palettized, x = input_49)[name = string("b_5")]; + tensor c_5 = silu(x = input_51)[name = string("c_5")]; + tensor input_53 = mul(x = c_5, y = b_5)[name = string("input_53")]; + string e_5_pad_type_0 = const()[name = string("e_5_pad_type_0"), val = string("valid")]; + tensor e_5_strides_0 = const()[name = string("e_5_strides_0"), val = tensor([1, 1])]; + tensor e_5_pad_0 = const()[name = string("e_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_5_dilations_0 = const()[name = string("e_5_dilations_0"), val = tensor([1, 1])]; + int32 e_5_groups_0 = const()[name = string("e_5_groups_0"), val = int32(1)]; + tensor e_5 = conv(dilations = e_5_dilations_0, groups = e_5_groups_0, pad = e_5_pad_0, pad_type = e_5_pad_type_0, strides = e_5_strides_0, weight = model_model_layers_2_mlp_down_proj_weight_palettized, x = input_53)[name = string("e_5")]; + tensor var_2273_axes_0 = const()[name = string("op_2273_axes_0"), val = tensor([2])]; + tensor var_2273 = squeeze(axes = var_2273_axes_0, x = e_5)[name = string("op_2273")]; + tensor var_2274 = const()[name = string("op_2274"), val = tensor([0, 2, 1])]; + tensor var_2275 = transpose(perm = var_2274, x = var_2273)[name = string("transpose_66")]; + tensor hidden_states_31_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = var_2275)[name = string("hidden_states_31_cast_fp16")]; + int32 var_2287 = const()[name = string("op_2287"), val = int32(-1)]; + fp16 const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2289_cast_fp16 = mul(x = hidden_states_31_cast_fp16, y = const_90_promoted_to_fp16)[name = string("op_2289_cast_fp16")]; + bool input_55_interleave_0 = const()[name = string("input_55_interleave_0"), val = bool(false)]; + tensor input_55_cast_fp16 = concat(axis = var_2287, interleave = input_55_interleave_0, values = (hidden_states_31_cast_fp16, var_2289_cast_fp16))[name = string("input_55_cast_fp16")]; + tensor normed_49_axes_0 = const()[name = string("normed_49_axes_0"), val = tensor([-1])]; + fp16 var_2284_to_fp16 = const()[name = string("op_2284_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_49_cast_fp16 = layer_norm(axes = normed_49_axes_0, epsilon = var_2284_to_fp16, x = input_55_cast_fp16)[name = string("normed_49_cast_fp16")]; + tensor normed_51_begin_0 = const()[name = string("normed_51_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_51_end_0 = const()[name = string("normed_51_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_51_end_mask_0 = const()[name = string("normed_51_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_51_cast_fp16 = slice_by_index(begin = normed_51_begin_0, end = normed_51_end_0, end_mask = normed_51_end_mask_0, x = normed_49_cast_fp16)[name = string("normed_51_cast_fp16")]; + tensor const_93_promoted_to_fp16 = const()[name = string("const_93_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676500800)))]; + tensor hidden_states_33_cast_fp16 = mul(x = normed_51_cast_fp16, y = const_93_promoted_to_fp16)[name = string("hidden_states_33_cast_fp16")]; + tensor var_2306 = const()[name = string("op_2306"), val = tensor([0, 2, 1])]; + tensor var_2309_axes_0 = const()[name = string("op_2309_axes_0"), val = tensor([2])]; + tensor var_2307_cast_fp16 = transpose(perm = var_2306, x = hidden_states_33_cast_fp16)[name = string("transpose_65")]; + tensor var_2309_cast_fp16 = expand_dims(axes = var_2309_axes_0, x = var_2307_cast_fp16)[name = string("op_2309_cast_fp16")]; + string var_2325_pad_type_0 = const()[name = string("op_2325_pad_type_0"), val = string("valid")]; + tensor var_2325_strides_0 = const()[name = string("op_2325_strides_0"), val = tensor([1, 1])]; + tensor var_2325_pad_0 = const()[name = string("op_2325_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2325_dilations_0 = const()[name = string("op_2325_dilations_0"), val = tensor([1, 1])]; + int32 var_2325_groups_0 = const()[name = string("op_2325_groups_0"), val = int32(1)]; + tensor var_2325 = conv(dilations = var_2325_dilations_0, groups = var_2325_groups_0, pad = var_2325_pad_0, pad_type = var_2325_pad_type_0, strides = var_2325_strides_0, weight = model_model_layers_3_self_attn_q_proj_weight_palettized, x = var_2309_cast_fp16)[name = string("op_2325")]; + tensor var_2330 = const()[name = string("op_2330"), val = tensor([1, 16, 1, 128])]; + tensor var_2331 = reshape(shape = var_2330, x = var_2325)[name = string("op_2331")]; + string var_2347_pad_type_0 = const()[name = string("op_2347_pad_type_0"), val = string("valid")]; + tensor var_2347_strides_0 = const()[name = string("op_2347_strides_0"), val = tensor([1, 1])]; + tensor var_2347_pad_0 = const()[name = string("op_2347_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2347_dilations_0 = const()[name = string("op_2347_dilations_0"), val = tensor([1, 1])]; + int32 var_2347_groups_0 = const()[name = string("op_2347_groups_0"), val = int32(1)]; + tensor var_2347 = conv(dilations = var_2347_dilations_0, groups = var_2347_groups_0, pad = var_2347_pad_0, pad_type = var_2347_pad_type_0, strides = var_2347_strides_0, weight = model_model_layers_3_self_attn_k_proj_weight_palettized, x = var_2309_cast_fp16)[name = string("op_2347")]; + tensor var_2352 = const()[name = string("op_2352"), val = tensor([1, 8, 1, 128])]; + tensor var_2353 = reshape(shape = var_2352, x = var_2347)[name = string("op_2353")]; + string var_2369_pad_type_0 = const()[name = string("op_2369_pad_type_0"), val = string("valid")]; + tensor var_2369_strides_0 = const()[name = string("op_2369_strides_0"), val = tensor([1, 1])]; + tensor var_2369_pad_0 = const()[name = string("op_2369_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2369_dilations_0 = const()[name = string("op_2369_dilations_0"), val = tensor([1, 1])]; + int32 var_2369_groups_0 = const()[name = string("op_2369_groups_0"), val = int32(1)]; + tensor var_2369 = conv(dilations = var_2369_dilations_0, groups = var_2369_groups_0, pad = var_2369_pad_0, pad_type = var_2369_pad_type_0, strides = var_2369_strides_0, weight = model_model_layers_3_self_attn_v_proj_weight_palettized, x = var_2309_cast_fp16)[name = string("op_2369")]; + tensor var_2374 = const()[name = string("op_2374"), val = tensor([1, 8, 1, 128])]; + tensor var_2375 = reshape(shape = var_2374, x = var_2369)[name = string("op_2375")]; + int32 var_2390 = const()[name = string("op_2390"), val = int32(-1)]; + fp16 const_94_promoted = const()[name = string("const_94_promoted"), val = fp16(-0x1p+0)]; + tensor var_2392 = mul(x = var_2331, y = const_94_promoted)[name = string("op_2392")]; + bool input_59_interleave_0 = const()[name = string("input_59_interleave_0"), val = bool(false)]; + tensor input_59 = concat(axis = var_2390, interleave = input_59_interleave_0, values = (var_2331, var_2392))[name = string("input_59")]; + tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; + fp16 var_2387_to_fp16 = const()[name = string("op_2387_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_2387_to_fp16, x = input_59)[name = string("normed_53_cast_fp16")]; + tensor normed_55_begin_0 = const()[name = string("normed_55_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_55_end_0 = const()[name = string("normed_55_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_55_end_mask_0 = const()[name = string("normed_55_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_55 = slice_by_index(begin = normed_55_begin_0, end = normed_55_end_0, end_mask = normed_55_end_mask_0, x = normed_53_cast_fp16)[name = string("normed_55")]; + tensor const_97 = const()[name = string("const_97"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676504960)))]; + tensor q_7 = mul(x = normed_55, y = const_97)[name = string("q_7")]; + int32 var_2415 = const()[name = string("op_2415"), val = int32(-1)]; + fp16 const_98_promoted = const()[name = string("const_98_promoted"), val = fp16(-0x1p+0)]; + tensor var_2417 = mul(x = var_2353, y = const_98_promoted)[name = string("op_2417")]; + bool input_61_interleave_0 = const()[name = string("input_61_interleave_0"), val = bool(false)]; + tensor input_61 = concat(axis = var_2415, interleave = input_61_interleave_0, values = (var_2353, var_2417))[name = string("input_61")]; + tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; + fp16 var_2412_to_fp16 = const()[name = string("op_2412_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_2412_to_fp16, x = input_61)[name = string("normed_57_cast_fp16")]; + tensor normed_59_begin_0 = const()[name = string("normed_59_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_59_end_0 = const()[name = string("normed_59_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_59_end_mask_0 = const()[name = string("normed_59_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_59 = slice_by_index(begin = normed_59_begin_0, end = normed_59_end_0, end_mask = normed_59_end_mask_0, x = normed_57_cast_fp16)[name = string("normed_59")]; + tensor const_101 = const()[name = string("const_101"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676505280)))]; + tensor k_7 = mul(x = normed_59, y = const_101)[name = string("k_7")]; + tensor var_2431 = mul(x = q_7, y = cos_1_cast_fp16)[name = string("op_2431")]; + tensor x1_13_begin_0 = const()[name = string("x1_13_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_13_end_0 = const()[name = string("x1_13_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_13_end_mask_0 = const()[name = string("x1_13_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_13 = slice_by_index(begin = x1_13_begin_0, end = x1_13_end_0, end_mask = x1_13_end_mask_0, x = q_7)[name = string("x1_13")]; + tensor x2_13_begin_0 = const()[name = string("x2_13_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_13_end_0 = const()[name = string("x2_13_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_13_end_mask_0 = const()[name = string("x2_13_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_13 = slice_by_index(begin = x2_13_begin_0, end = x2_13_end_0, end_mask = x2_13_end_mask_0, x = q_7)[name = string("x2_13")]; + fp16 const_104_promoted = const()[name = string("const_104_promoted"), val = fp16(-0x1p+0)]; + tensor var_2452 = mul(x = x2_13, y = const_104_promoted)[name = string("op_2452")]; + int32 var_2454 = const()[name = string("op_2454"), val = int32(-1)]; + bool var_2455_interleave_0 = const()[name = string("op_2455_interleave_0"), val = bool(false)]; + tensor var_2455 = concat(axis = var_2454, interleave = var_2455_interleave_0, values = (var_2452, x1_13))[name = string("op_2455")]; + tensor var_2456 = mul(x = var_2455, y = sin_1_cast_fp16)[name = string("op_2456")]; + tensor query_states_13 = add(x = var_2431, y = var_2456)[name = string("query_states_13")]; + tensor var_2459 = mul(x = k_7, y = cos_1_cast_fp16)[name = string("op_2459")]; + tensor x1_15_begin_0 = const()[name = string("x1_15_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_15_end_0 = const()[name = string("x1_15_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_15_end_mask_0 = const()[name = string("x1_15_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_15 = slice_by_index(begin = x1_15_begin_0, end = x1_15_end_0, end_mask = x1_15_end_mask_0, x = k_7)[name = string("x1_15")]; + tensor x2_15_begin_0 = const()[name = string("x2_15_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_15_end_0 = const()[name = string("x2_15_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_15_end_mask_0 = const()[name = string("x2_15_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_15 = slice_by_index(begin = x2_15_begin_0, end = x2_15_end_0, end_mask = x2_15_end_mask_0, x = k_7)[name = string("x2_15")]; + fp16 const_107_promoted = const()[name = string("const_107_promoted"), val = fp16(-0x1p+0)]; + tensor var_2480 = mul(x = x2_15, y = const_107_promoted)[name = string("op_2480")]; + int32 var_2482 = const()[name = string("op_2482"), val = int32(-1)]; + bool var_2483_interleave_0 = const()[name = string("op_2483_interleave_0"), val = bool(false)]; + tensor var_2483 = concat(axis = var_2482, interleave = var_2483_interleave_0, values = (var_2480, x1_15))[name = string("op_2483")]; + tensor var_2484 = mul(x = var_2483, y = sin_1_cast_fp16)[name = string("op_2484")]; + tensor key_states_13 = add(x = var_2459, y = var_2484)[name = string("key_states_13")]; + tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([3])]; + tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; + tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; + tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([4])]; + int32 concat_26_axis_0 = const()[name = string("concat_26_axis_0"), val = int32(0)]; + bool concat_26_interleave_0 = const()[name = string("concat_26_interleave_0"), val = bool(false)]; + tensor concat_26 = concat(axis = concat_26_axis_0, interleave = concat_26_interleave_0, values = (expand_dims_36, expand_dims_37, current_pos, expand_dims_39))[name = string("concat_26")]; + tensor concat_27_values1_0 = const()[name = string("concat_27_values1_0"), val = tensor([0])]; + tensor concat_27_values3_0 = const()[name = string("concat_27_values3_0"), val = tensor([0])]; + int32 concat_27_axis_0 = const()[name = string("concat_27_axis_0"), val = int32(0)]; + bool concat_27_interleave_0 = const()[name = string("concat_27_interleave_0"), val = bool(false)]; + tensor concat_27 = concat(axis = concat_27_axis_0, interleave = concat_27_interleave_0, values = (expand_dims_40, concat_27_values1_0, var_1001, concat_27_values3_0))[name = string("concat_27")]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_26, begin_mask = model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0, end = concat_27, end_mask = model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_7_stride_0, update = key_states_13, x = coreml_update_state_33)[name = string("model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_62_write_state")]; + tensor coreml_update_state_34 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_62")]; + tensor expand_dims_42 = const()[name = string("expand_dims_42"), val = tensor([31])]; + tensor expand_dims_43 = const()[name = string("expand_dims_43"), val = tensor([0])]; + tensor expand_dims_45 = const()[name = string("expand_dims_45"), val = tensor([0])]; + tensor expand_dims_46 = const()[name = string("expand_dims_46"), val = tensor([32])]; + int32 concat_30_axis_0 = const()[name = string("concat_30_axis_0"), val = int32(0)]; + bool concat_30_interleave_0 = const()[name = string("concat_30_interleave_0"), val = bool(false)]; + tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (expand_dims_42, expand_dims_43, current_pos, expand_dims_45))[name = string("concat_30")]; + tensor concat_31_values1_0 = const()[name = string("concat_31_values1_0"), val = tensor([0])]; + tensor concat_31_values3_0 = const()[name = string("concat_31_values3_0"), val = tensor([0])]; + int32 concat_31_axis_0 = const()[name = string("concat_31_axis_0"), val = int32(0)]; + bool concat_31_interleave_0 = const()[name = string("concat_31_interleave_0"), val = bool(false)]; + tensor concat_31 = concat(axis = concat_31_axis_0, interleave = concat_31_interleave_0, values = (expand_dims_46, concat_31_values1_0, var_1001, concat_31_values3_0))[name = string("concat_31")]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_30, begin_mask = model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0, end = concat_31, end_mask = model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_8_stride_0, update = var_2375, x = coreml_update_state_34)[name = string("model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_63_write_state")]; + tensor coreml_update_state_35 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_63")]; + tensor var_2539_begin_0 = const()[name = string("op_2539_begin_0"), val = tensor([3, 0, 0, 0])]; + tensor var_2539_end_0 = const()[name = string("op_2539_end_0"), val = tensor([4, 8, 1024, 128])]; + tensor var_2539_end_mask_0 = const()[name = string("op_2539_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2539_cast_fp16 = slice_by_index(begin = var_2539_begin_0, end = var_2539_end_0, end_mask = var_2539_end_mask_0, x = coreml_update_state_35)[name = string("op_2539_cast_fp16")]; + tensor K_layer_cache_7_axes_0 = const()[name = string("K_layer_cache_7_axes_0"), val = tensor([0])]; + tensor K_layer_cache_7_cast_fp16 = squeeze(axes = K_layer_cache_7_axes_0, x = var_2539_cast_fp16)[name = string("K_layer_cache_7_cast_fp16")]; + tensor var_2546_begin_0 = const()[name = string("op_2546_begin_0"), val = tensor([31, 0, 0, 0])]; + tensor var_2546_end_0 = const()[name = string("op_2546_end_0"), val = tensor([32, 8, 1024, 128])]; + tensor var_2546_end_mask_0 = const()[name = string("op_2546_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2546_cast_fp16 = slice_by_index(begin = var_2546_begin_0, end = var_2546_end_0, end_mask = var_2546_end_mask_0, x = coreml_update_state_35)[name = string("op_2546_cast_fp16")]; + tensor V_layer_cache_7_axes_0 = const()[name = string("V_layer_cache_7_axes_0"), val = tensor([0])]; + tensor V_layer_cache_7_cast_fp16 = squeeze(axes = V_layer_cache_7_axes_0, x = var_2546_cast_fp16)[name = string("V_layer_cache_7_cast_fp16")]; + tensor x_51_axes_0 = const()[name = string("x_51_axes_0"), val = tensor([1])]; + tensor x_51_cast_fp16 = expand_dims(axes = x_51_axes_0, x = K_layer_cache_7_cast_fp16)[name = string("x_51_cast_fp16")]; + tensor var_2583 = const()[name = string("op_2583"), val = tensor([1, 2, 1, 1])]; + tensor x_53_cast_fp16 = tile(reps = var_2583, x = x_51_cast_fp16)[name = string("x_53_cast_fp16")]; + tensor var_2595 = const()[name = string("op_2595"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_15_cast_fp16 = reshape(shape = var_2595, x = x_53_cast_fp16)[name = string("key_states_15_cast_fp16")]; + tensor x_57_axes_0 = const()[name = string("x_57_axes_0"), val = tensor([1])]; + tensor x_57_cast_fp16 = expand_dims(axes = x_57_axes_0, x = V_layer_cache_7_cast_fp16)[name = string("x_57_cast_fp16")]; + tensor var_2603 = const()[name = string("op_2603"), val = tensor([1, 2, 1, 1])]; + tensor x_59_cast_fp16 = tile(reps = var_2603, x = x_57_cast_fp16)[name = string("x_59_cast_fp16")]; + tensor var_2615 = const()[name = string("op_2615"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_21_cast_fp16 = reshape(shape = var_2615, x = x_59_cast_fp16)[name = string("value_states_21_cast_fp16")]; + bool var_2630_transpose_x_1 = const()[name = string("op_2630_transpose_x_1"), val = bool(false)]; + bool var_2630_transpose_y_1 = const()[name = string("op_2630_transpose_y_1"), val = bool(true)]; + tensor var_2630 = matmul(transpose_x = var_2630_transpose_x_1, transpose_y = var_2630_transpose_y_1, x = query_states_13, y = key_states_15_cast_fp16)[name = string("op_2630")]; + fp16 var_2631_to_fp16 = const()[name = string("op_2631_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_19_cast_fp16 = mul(x = var_2630, y = var_2631_to_fp16)[name = string("attn_weights_19_cast_fp16")]; + tensor attn_weights_21_cast_fp16 = add(x = attn_weights_19_cast_fp16, y = causal_mask)[name = string("attn_weights_21_cast_fp16")]; + int32 var_2666 = const()[name = string("op_2666"), val = int32(-1)]; + tensor attn_weights_23_cast_fp16 = softmax(axis = var_2666, x = attn_weights_21_cast_fp16)[name = string("attn_weights_23_cast_fp16")]; + bool attn_output_31_transpose_x_0 = const()[name = string("attn_output_31_transpose_x_0"), val = bool(false)]; + bool attn_output_31_transpose_y_0 = const()[name = string("attn_output_31_transpose_y_0"), val = bool(false)]; + tensor attn_output_31_cast_fp16 = matmul(transpose_x = attn_output_31_transpose_x_0, transpose_y = attn_output_31_transpose_y_0, x = attn_weights_23_cast_fp16, y = value_states_21_cast_fp16)[name = string("attn_output_31_cast_fp16")]; + tensor var_2677_perm_0 = const()[name = string("op_2677_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_2681 = const()[name = string("op_2681"), val = tensor([1, 1, 2048])]; + tensor var_2677_cast_fp16 = transpose(perm = var_2677_perm_0, x = attn_output_31_cast_fp16)[name = string("transpose_64")]; + tensor attn_output_35_cast_fp16 = reshape(shape = var_2681, x = var_2677_cast_fp16)[name = string("attn_output_35_cast_fp16")]; + tensor var_2686 = const()[name = string("op_2686"), val = tensor([0, 2, 1])]; + string var_2702_pad_type_0 = const()[name = string("op_2702_pad_type_0"), val = string("valid")]; + int32 var_2702_groups_0 = const()[name = string("op_2702_groups_0"), val = int32(1)]; + tensor var_2702_strides_0 = const()[name = string("op_2702_strides_0"), val = tensor([1])]; + tensor var_2702_pad_0 = const()[name = string("op_2702_pad_0"), val = tensor([0, 0])]; + tensor var_2702_dilations_0 = const()[name = string("op_2702_dilations_0"), val = tensor([1])]; + tensor squeeze_3_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676505600))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680699968))))[name = string("squeeze_3_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_2687_cast_fp16 = transpose(perm = var_2686, x = attn_output_35_cast_fp16)[name = string("transpose_63")]; + tensor var_2702_cast_fp16 = conv(dilations = var_2702_dilations_0, groups = var_2702_groups_0, pad = var_2702_pad_0, pad_type = var_2702_pad_type_0, strides = var_2702_strides_0, weight = squeeze_3_cast_fp16_to_fp32_to_fp16_palettized, x = var_2687_cast_fp16)[name = string("op_2702_cast_fp16")]; + tensor var_2706 = const()[name = string("op_2706"), val = tensor([0, 2, 1])]; + tensor attn_output_39_cast_fp16 = transpose(perm = var_2706, x = var_2702_cast_fp16)[name = string("transpose_62")]; + tensor hidden_states_39_cast_fp16 = add(x = hidden_states_31_cast_fp16, y = attn_output_39_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; + int32 var_2719 = const()[name = string("op_2719"), val = int32(-1)]; + fp16 const_116_promoted_to_fp16 = const()[name = string("const_116_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2721_cast_fp16 = mul(x = hidden_states_39_cast_fp16, y = const_116_promoted_to_fp16)[name = string("op_2721_cast_fp16")]; + bool input_65_interleave_0 = const()[name = string("input_65_interleave_0"), val = bool(false)]; + tensor input_65_cast_fp16 = concat(axis = var_2719, interleave = input_65_interleave_0, values = (hidden_states_39_cast_fp16, var_2721_cast_fp16))[name = string("input_65_cast_fp16")]; + tensor normed_61_axes_0 = const()[name = string("normed_61_axes_0"), val = tensor([-1])]; + fp16 var_2716_to_fp16 = const()[name = string("op_2716_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_61_cast_fp16 = layer_norm(axes = normed_61_axes_0, epsilon = var_2716_to_fp16, x = input_65_cast_fp16)[name = string("normed_61_cast_fp16")]; + tensor normed_63_begin_0 = const()[name = string("normed_63_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_63_end_0 = const()[name = string("normed_63_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_63_end_mask_0 = const()[name = string("normed_63_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_63_cast_fp16 = slice_by_index(begin = normed_63_begin_0, end = normed_63_end_0, end_mask = normed_63_end_mask_0, x = normed_61_cast_fp16)[name = string("normed_63_cast_fp16")]; + tensor const_119_promoted_to_fp16 = const()[name = string("const_119_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680831104)))]; + tensor x_61_cast_fp16 = mul(x = normed_63_cast_fp16, y = const_119_promoted_to_fp16)[name = string("x_61_cast_fp16")]; + tensor var_2746 = const()[name = string("op_2746"), val = tensor([0, 2, 1])]; + tensor input_67_axes_0 = const()[name = string("input_67_axes_0"), val = tensor([2])]; + tensor var_2747 = transpose(perm = var_2746, x = x_61_cast_fp16)[name = string("transpose_61")]; + tensor input_67 = expand_dims(axes = input_67_axes_0, x = var_2747)[name = string("input_67")]; + string input_69_pad_type_0 = const()[name = string("input_69_pad_type_0"), val = string("valid")]; + tensor input_69_strides_0 = const()[name = string("input_69_strides_0"), val = tensor([1, 1])]; + tensor input_69_pad_0 = const()[name = string("input_69_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_69_dilations_0 = const()[name = string("input_69_dilations_0"), val = tensor([1, 1])]; + int32 input_69_groups_0 = const()[name = string("input_69_groups_0"), val = int32(1)]; + tensor input_69 = conv(dilations = input_69_dilations_0, groups = input_69_groups_0, pad = input_69_pad_0, pad_type = input_69_pad_type_0, strides = input_69_strides_0, weight = model_model_layers_3_mlp_gate_proj_weight_palettized, x = input_67)[name = string("input_69")]; + string b_7_pad_type_0 = const()[name = string("b_7_pad_type_0"), val = string("valid")]; + tensor b_7_strides_0 = const()[name = string("b_7_strides_0"), val = tensor([1, 1])]; + tensor b_7_pad_0 = const()[name = string("b_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_7_dilations_0 = const()[name = string("b_7_dilations_0"), val = tensor([1, 1])]; + int32 b_7_groups_0 = const()[name = string("b_7_groups_0"), val = int32(1)]; + tensor b_7 = conv(dilations = b_7_dilations_0, groups = b_7_groups_0, pad = b_7_pad_0, pad_type = b_7_pad_type_0, strides = b_7_strides_0, weight = model_model_layers_3_mlp_up_proj_weight_palettized, x = input_67)[name = string("b_7")]; + tensor c_7 = silu(x = input_69)[name = string("c_7")]; + tensor input_71 = mul(x = c_7, y = b_7)[name = string("input_71")]; + string e_7_pad_type_0 = const()[name = string("e_7_pad_type_0"), val = string("valid")]; + tensor e_7_strides_0 = const()[name = string("e_7_strides_0"), val = tensor([1, 1])]; + tensor e_7_pad_0 = const()[name = string("e_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_7_dilations_0 = const()[name = string("e_7_dilations_0"), val = tensor([1, 1])]; + int32 e_7_groups_0 = const()[name = string("e_7_groups_0"), val = int32(1)]; + tensor e_7 = conv(dilations = e_7_dilations_0, groups = e_7_groups_0, pad = e_7_pad_0, pad_type = e_7_pad_type_0, strides = e_7_strides_0, weight = model_model_layers_3_mlp_down_proj_weight_palettized, x = input_71)[name = string("e_7")]; + tensor var_2769_axes_0 = const()[name = string("op_2769_axes_0"), val = tensor([2])]; + tensor var_2769 = squeeze(axes = var_2769_axes_0, x = e_7)[name = string("op_2769")]; + tensor var_2770 = const()[name = string("op_2770"), val = tensor([0, 2, 1])]; + tensor var_2771 = transpose(perm = var_2770, x = var_2769)[name = string("transpose_60")]; + tensor hidden_states_41_cast_fp16 = add(x = hidden_states_39_cast_fp16, y = var_2771)[name = string("hidden_states_41_cast_fp16")]; + int32 var_2783 = const()[name = string("op_2783"), val = int32(-1)]; + fp16 const_120_promoted_to_fp16 = const()[name = string("const_120_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2785_cast_fp16 = mul(x = hidden_states_41_cast_fp16, y = const_120_promoted_to_fp16)[name = string("op_2785_cast_fp16")]; + bool input_73_interleave_0 = const()[name = string("input_73_interleave_0"), val = bool(false)]; + tensor input_73_cast_fp16 = concat(axis = var_2783, interleave = input_73_interleave_0, values = (hidden_states_41_cast_fp16, var_2785_cast_fp16))[name = string("input_73_cast_fp16")]; + tensor normed_65_axes_0 = const()[name = string("normed_65_axes_0"), val = tensor([-1])]; + fp16 var_2780_to_fp16 = const()[name = string("op_2780_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_65_cast_fp16 = layer_norm(axes = normed_65_axes_0, epsilon = var_2780_to_fp16, x = input_73_cast_fp16)[name = string("normed_65_cast_fp16")]; + tensor normed_67_begin_0 = const()[name = string("normed_67_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_67_end_0 = const()[name = string("normed_67_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_67_end_mask_0 = const()[name = string("normed_67_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_67_cast_fp16 = slice_by_index(begin = normed_67_begin_0, end = normed_67_end_0, end_mask = normed_67_end_mask_0, x = normed_65_cast_fp16)[name = string("normed_67_cast_fp16")]; + tensor const_123_promoted_to_fp16 = const()[name = string("const_123_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680835264)))]; + tensor hidden_states_43_cast_fp16 = mul(x = normed_67_cast_fp16, y = const_123_promoted_to_fp16)[name = string("hidden_states_43_cast_fp16")]; + tensor var_2802 = const()[name = string("op_2802"), val = tensor([0, 2, 1])]; + tensor var_2805_axes_0 = const()[name = string("op_2805_axes_0"), val = tensor([2])]; + tensor var_2803_cast_fp16 = transpose(perm = var_2802, x = hidden_states_43_cast_fp16)[name = string("transpose_59")]; + tensor var_2805_cast_fp16 = expand_dims(axes = var_2805_axes_0, x = var_2803_cast_fp16)[name = string("op_2805_cast_fp16")]; + string var_2821_pad_type_0 = const()[name = string("op_2821_pad_type_0"), val = string("valid")]; + tensor var_2821_strides_0 = const()[name = string("op_2821_strides_0"), val = tensor([1, 1])]; + tensor var_2821_pad_0 = const()[name = string("op_2821_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2821_dilations_0 = const()[name = string("op_2821_dilations_0"), val = tensor([1, 1])]; + int32 var_2821_groups_0 = const()[name = string("op_2821_groups_0"), val = int32(1)]; + tensor var_2821 = conv(dilations = var_2821_dilations_0, groups = var_2821_groups_0, pad = var_2821_pad_0, pad_type = var_2821_pad_type_0, strides = var_2821_strides_0, weight = model_model_layers_4_self_attn_q_proj_weight_palettized, x = var_2805_cast_fp16)[name = string("op_2821")]; + tensor var_2826 = const()[name = string("op_2826"), val = tensor([1, 16, 1, 128])]; + tensor var_2827 = reshape(shape = var_2826, x = var_2821)[name = string("op_2827")]; + string var_2843_pad_type_0 = const()[name = string("op_2843_pad_type_0"), val = string("valid")]; + tensor var_2843_strides_0 = const()[name = string("op_2843_strides_0"), val = tensor([1, 1])]; + tensor var_2843_pad_0 = const()[name = string("op_2843_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2843_dilations_0 = const()[name = string("op_2843_dilations_0"), val = tensor([1, 1])]; + int32 var_2843_groups_0 = const()[name = string("op_2843_groups_0"), val = int32(1)]; + tensor var_2843 = conv(dilations = var_2843_dilations_0, groups = var_2843_groups_0, pad = var_2843_pad_0, pad_type = var_2843_pad_type_0, strides = var_2843_strides_0, weight = model_model_layers_4_self_attn_k_proj_weight_palettized, x = var_2805_cast_fp16)[name = string("op_2843")]; + tensor var_2848 = const()[name = string("op_2848"), val = tensor([1, 8, 1, 128])]; + tensor var_2849 = reshape(shape = var_2848, x = var_2843)[name = string("op_2849")]; + string var_2865_pad_type_0 = const()[name = string("op_2865_pad_type_0"), val = string("valid")]; + tensor var_2865_strides_0 = const()[name = string("op_2865_strides_0"), val = tensor([1, 1])]; + tensor var_2865_pad_0 = const()[name = string("op_2865_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2865_dilations_0 = const()[name = string("op_2865_dilations_0"), val = tensor([1, 1])]; + int32 var_2865_groups_0 = const()[name = string("op_2865_groups_0"), val = int32(1)]; + tensor var_2865 = conv(dilations = var_2865_dilations_0, groups = var_2865_groups_0, pad = var_2865_pad_0, pad_type = var_2865_pad_type_0, strides = var_2865_strides_0, weight = model_model_layers_4_self_attn_v_proj_weight_palettized, x = var_2805_cast_fp16)[name = string("op_2865")]; + tensor var_2870 = const()[name = string("op_2870"), val = tensor([1, 8, 1, 128])]; + tensor var_2871 = reshape(shape = var_2870, x = var_2865)[name = string("op_2871")]; + int32 var_2886 = const()[name = string("op_2886"), val = int32(-1)]; + fp16 const_124_promoted = const()[name = string("const_124_promoted"), val = fp16(-0x1p+0)]; + tensor var_2888 = mul(x = var_2827, y = const_124_promoted)[name = string("op_2888")]; + bool input_77_interleave_0 = const()[name = string("input_77_interleave_0"), val = bool(false)]; + tensor input_77 = concat(axis = var_2886, interleave = input_77_interleave_0, values = (var_2827, var_2888))[name = string("input_77")]; + tensor normed_69_axes_0 = const()[name = string("normed_69_axes_0"), val = tensor([-1])]; + fp16 var_2883_to_fp16 = const()[name = string("op_2883_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_69_cast_fp16 = layer_norm(axes = normed_69_axes_0, epsilon = var_2883_to_fp16, x = input_77)[name = string("normed_69_cast_fp16")]; + tensor normed_71_begin_0 = const()[name = string("normed_71_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_71_end_0 = const()[name = string("normed_71_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_71_end_mask_0 = const()[name = string("normed_71_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_71 = slice_by_index(begin = normed_71_begin_0, end = normed_71_end_0, end_mask = normed_71_end_mask_0, x = normed_69_cast_fp16)[name = string("normed_71")]; + tensor const_127 = const()[name = string("const_127"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680839424)))]; + tensor q_9 = mul(x = normed_71, y = const_127)[name = string("q_9")]; + int32 var_2911 = const()[name = string("op_2911"), val = int32(-1)]; + fp16 const_128_promoted = const()[name = string("const_128_promoted"), val = fp16(-0x1p+0)]; + tensor var_2913 = mul(x = var_2849, y = const_128_promoted)[name = string("op_2913")]; + bool input_79_interleave_0 = const()[name = string("input_79_interleave_0"), val = bool(false)]; + tensor input_79 = concat(axis = var_2911, interleave = input_79_interleave_0, values = (var_2849, var_2913))[name = string("input_79")]; + tensor normed_73_axes_0 = const()[name = string("normed_73_axes_0"), val = tensor([-1])]; + fp16 var_2908_to_fp16 = const()[name = string("op_2908_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_73_cast_fp16 = layer_norm(axes = normed_73_axes_0, epsilon = var_2908_to_fp16, x = input_79)[name = string("normed_73_cast_fp16")]; + tensor normed_75_begin_0 = const()[name = string("normed_75_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_75_end_0 = const()[name = string("normed_75_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_75_end_mask_0 = const()[name = string("normed_75_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_75 = slice_by_index(begin = normed_75_begin_0, end = normed_75_end_0, end_mask = normed_75_end_mask_0, x = normed_73_cast_fp16)[name = string("normed_75")]; + tensor const_131 = const()[name = string("const_131"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680839744)))]; + tensor k_9 = mul(x = normed_75, y = const_131)[name = string("k_9")]; + tensor var_2927 = mul(x = q_9, y = cos_1_cast_fp16)[name = string("op_2927")]; + tensor x1_17_begin_0 = const()[name = string("x1_17_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_17_end_0 = const()[name = string("x1_17_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_17_end_mask_0 = const()[name = string("x1_17_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_17 = slice_by_index(begin = x1_17_begin_0, end = x1_17_end_0, end_mask = x1_17_end_mask_0, x = q_9)[name = string("x1_17")]; + tensor x2_17_begin_0 = const()[name = string("x2_17_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_17_end_0 = const()[name = string("x2_17_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_17_end_mask_0 = const()[name = string("x2_17_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_17 = slice_by_index(begin = x2_17_begin_0, end = x2_17_end_0, end_mask = x2_17_end_mask_0, x = q_9)[name = string("x2_17")]; + fp16 const_134_promoted = const()[name = string("const_134_promoted"), val = fp16(-0x1p+0)]; + tensor var_2948 = mul(x = x2_17, y = const_134_promoted)[name = string("op_2948")]; + int32 var_2950 = const()[name = string("op_2950"), val = int32(-1)]; + bool var_2951_interleave_0 = const()[name = string("op_2951_interleave_0"), val = bool(false)]; + tensor var_2951 = concat(axis = var_2950, interleave = var_2951_interleave_0, values = (var_2948, x1_17))[name = string("op_2951")]; + tensor var_2952 = mul(x = var_2951, y = sin_1_cast_fp16)[name = string("op_2952")]; + tensor query_states_17 = add(x = var_2927, y = var_2952)[name = string("query_states_17")]; + tensor var_2955 = mul(x = k_9, y = cos_1_cast_fp16)[name = string("op_2955")]; + tensor x1_19_begin_0 = const()[name = string("x1_19_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_19_end_0 = const()[name = string("x1_19_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_19_end_mask_0 = const()[name = string("x1_19_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_19 = slice_by_index(begin = x1_19_begin_0, end = x1_19_end_0, end_mask = x1_19_end_mask_0, x = k_9)[name = string("x1_19")]; + tensor x2_19_begin_0 = const()[name = string("x2_19_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_19_end_0 = const()[name = string("x2_19_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_19_end_mask_0 = const()[name = string("x2_19_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_19 = slice_by_index(begin = x2_19_begin_0, end = x2_19_end_0, end_mask = x2_19_end_mask_0, x = k_9)[name = string("x2_19")]; + fp16 const_137_promoted = const()[name = string("const_137_promoted"), val = fp16(-0x1p+0)]; + tensor var_2976 = mul(x = x2_19, y = const_137_promoted)[name = string("op_2976")]; + int32 var_2978 = const()[name = string("op_2978"), val = int32(-1)]; + bool var_2979_interleave_0 = const()[name = string("op_2979_interleave_0"), val = bool(false)]; + tensor var_2979 = concat(axis = var_2978, interleave = var_2979_interleave_0, values = (var_2976, x1_19))[name = string("op_2979")]; + tensor var_2980 = mul(x = var_2979, y = sin_1_cast_fp16)[name = string("op_2980")]; + tensor key_states_17 = add(x = var_2955, y = var_2980)[name = string("key_states_17")]; + tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([4])]; + tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; + tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; + tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([5])]; + int32 concat_34_axis_0 = const()[name = string("concat_34_axis_0"), val = int32(0)]; + bool concat_34_interleave_0 = const()[name = string("concat_34_interleave_0"), val = bool(false)]; + tensor concat_34 = concat(axis = concat_34_axis_0, interleave = concat_34_interleave_0, values = (expand_dims_48, expand_dims_49, current_pos, expand_dims_51))[name = string("concat_34")]; + tensor concat_35_values1_0 = const()[name = string("concat_35_values1_0"), val = tensor([0])]; + tensor concat_35_values3_0 = const()[name = string("concat_35_values3_0"), val = tensor([0])]; + int32 concat_35_axis_0 = const()[name = string("concat_35_axis_0"), val = int32(0)]; + bool concat_35_interleave_0 = const()[name = string("concat_35_interleave_0"), val = bool(false)]; + tensor concat_35 = concat(axis = concat_35_axis_0, interleave = concat_35_interleave_0, values = (expand_dims_52, concat_35_values1_0, var_1001, concat_35_values3_0))[name = string("concat_35")]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_34, begin_mask = model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0, end = concat_35, end_mask = model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_9_stride_0, update = key_states_17, x = coreml_update_state_35)[name = string("model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_64_write_state")]; + tensor coreml_update_state_36 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_64")]; + tensor expand_dims_54 = const()[name = string("expand_dims_54"), val = tensor([32])]; + tensor expand_dims_55 = const()[name = string("expand_dims_55"), val = tensor([0])]; + tensor expand_dims_57 = const()[name = string("expand_dims_57"), val = tensor([0])]; + tensor expand_dims_58 = const()[name = string("expand_dims_58"), val = tensor([33])]; + int32 concat_38_axis_0 = const()[name = string("concat_38_axis_0"), val = int32(0)]; + bool concat_38_interleave_0 = const()[name = string("concat_38_interleave_0"), val = bool(false)]; + tensor concat_38 = concat(axis = concat_38_axis_0, interleave = concat_38_interleave_0, values = (expand_dims_54, expand_dims_55, current_pos, expand_dims_57))[name = string("concat_38")]; + tensor concat_39_values1_0 = const()[name = string("concat_39_values1_0"), val = tensor([0])]; + tensor concat_39_values3_0 = const()[name = string("concat_39_values3_0"), val = tensor([0])]; + int32 concat_39_axis_0 = const()[name = string("concat_39_axis_0"), val = int32(0)]; + bool concat_39_interleave_0 = const()[name = string("concat_39_interleave_0"), val = bool(false)]; + tensor concat_39 = concat(axis = concat_39_axis_0, interleave = concat_39_interleave_0, values = (expand_dims_58, concat_39_values1_0, var_1001, concat_39_values3_0))[name = string("concat_39")]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_38, begin_mask = model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0, end = concat_39, end_mask = model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_10_stride_0, update = var_2871, x = coreml_update_state_36)[name = string("model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_65_write_state")]; + tensor coreml_update_state_37 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_65")]; + tensor var_3035_begin_0 = const()[name = string("op_3035_begin_0"), val = tensor([4, 0, 0, 0])]; + tensor var_3035_end_0 = const()[name = string("op_3035_end_0"), val = tensor([5, 8, 1024, 128])]; + tensor var_3035_end_mask_0 = const()[name = string("op_3035_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3035_cast_fp16 = slice_by_index(begin = var_3035_begin_0, end = var_3035_end_0, end_mask = var_3035_end_mask_0, x = coreml_update_state_37)[name = string("op_3035_cast_fp16")]; + tensor K_layer_cache_9_axes_0 = const()[name = string("K_layer_cache_9_axes_0"), val = tensor([0])]; + tensor K_layer_cache_9_cast_fp16 = squeeze(axes = K_layer_cache_9_axes_0, x = var_3035_cast_fp16)[name = string("K_layer_cache_9_cast_fp16")]; + tensor var_3042_begin_0 = const()[name = string("op_3042_begin_0"), val = tensor([32, 0, 0, 0])]; + tensor var_3042_end_0 = const()[name = string("op_3042_end_0"), val = tensor([33, 8, 1024, 128])]; + tensor var_3042_end_mask_0 = const()[name = string("op_3042_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3042_cast_fp16 = slice_by_index(begin = var_3042_begin_0, end = var_3042_end_0, end_mask = var_3042_end_mask_0, x = coreml_update_state_37)[name = string("op_3042_cast_fp16")]; + tensor V_layer_cache_9_axes_0 = const()[name = string("V_layer_cache_9_axes_0"), val = tensor([0])]; + tensor V_layer_cache_9_cast_fp16 = squeeze(axes = V_layer_cache_9_axes_0, x = var_3042_cast_fp16)[name = string("V_layer_cache_9_cast_fp16")]; + tensor x_67_axes_0 = const()[name = string("x_67_axes_0"), val = tensor([1])]; + tensor x_67_cast_fp16 = expand_dims(axes = x_67_axes_0, x = K_layer_cache_9_cast_fp16)[name = string("x_67_cast_fp16")]; + tensor var_3079 = const()[name = string("op_3079"), val = tensor([1, 2, 1, 1])]; + tensor x_69_cast_fp16 = tile(reps = var_3079, x = x_67_cast_fp16)[name = string("x_69_cast_fp16")]; + tensor var_3091 = const()[name = string("op_3091"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_19_cast_fp16 = reshape(shape = var_3091, x = x_69_cast_fp16)[name = string("key_states_19_cast_fp16")]; + tensor x_73_axes_0 = const()[name = string("x_73_axes_0"), val = tensor([1])]; + tensor x_73_cast_fp16 = expand_dims(axes = x_73_axes_0, x = V_layer_cache_9_cast_fp16)[name = string("x_73_cast_fp16")]; + tensor var_3099 = const()[name = string("op_3099"), val = tensor([1, 2, 1, 1])]; + tensor x_75_cast_fp16 = tile(reps = var_3099, x = x_73_cast_fp16)[name = string("x_75_cast_fp16")]; + tensor var_3111 = const()[name = string("op_3111"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_27_cast_fp16 = reshape(shape = var_3111, x = x_75_cast_fp16)[name = string("value_states_27_cast_fp16")]; + bool var_3126_transpose_x_1 = const()[name = string("op_3126_transpose_x_1"), val = bool(false)]; + bool var_3126_transpose_y_1 = const()[name = string("op_3126_transpose_y_1"), val = bool(true)]; + tensor var_3126 = matmul(transpose_x = var_3126_transpose_x_1, transpose_y = var_3126_transpose_y_1, x = query_states_17, y = key_states_19_cast_fp16)[name = string("op_3126")]; + fp16 var_3127_to_fp16 = const()[name = string("op_3127_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_25_cast_fp16 = mul(x = var_3126, y = var_3127_to_fp16)[name = string("attn_weights_25_cast_fp16")]; + tensor attn_weights_27_cast_fp16 = add(x = attn_weights_25_cast_fp16, y = causal_mask)[name = string("attn_weights_27_cast_fp16")]; + int32 var_3162 = const()[name = string("op_3162"), val = int32(-1)]; + tensor attn_weights_29_cast_fp16 = softmax(axis = var_3162, x = attn_weights_27_cast_fp16)[name = string("attn_weights_29_cast_fp16")]; + bool attn_output_41_transpose_x_0 = const()[name = string("attn_output_41_transpose_x_0"), val = bool(false)]; + bool attn_output_41_transpose_y_0 = const()[name = string("attn_output_41_transpose_y_0"), val = bool(false)]; + tensor attn_output_41_cast_fp16 = matmul(transpose_x = attn_output_41_transpose_x_0, transpose_y = attn_output_41_transpose_y_0, x = attn_weights_29_cast_fp16, y = value_states_27_cast_fp16)[name = string("attn_output_41_cast_fp16")]; + tensor var_3173_perm_0 = const()[name = string("op_3173_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_3177 = const()[name = string("op_3177"), val = tensor([1, 1, 2048])]; + tensor var_3173_cast_fp16 = transpose(perm = var_3173_perm_0, x = attn_output_41_cast_fp16)[name = string("transpose_58")]; + tensor attn_output_45_cast_fp16 = reshape(shape = var_3177, x = var_3173_cast_fp16)[name = string("attn_output_45_cast_fp16")]; + tensor var_3182 = const()[name = string("op_3182"), val = tensor([0, 2, 1])]; + string var_3198_pad_type_0 = const()[name = string("op_3198_pad_type_0"), val = string("valid")]; + int32 var_3198_groups_0 = const()[name = string("op_3198_groups_0"), val = int32(1)]; + tensor var_3198_strides_0 = const()[name = string("op_3198_strides_0"), val = tensor([1])]; + tensor var_3198_pad_0 = const()[name = string("op_3198_pad_0"), val = tensor([0, 0])]; + tensor var_3198_dilations_0 = const()[name = string("op_3198_dilations_0"), val = tensor([1])]; + tensor squeeze_4_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680840064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685034432))))[name = string("squeeze_4_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_3183_cast_fp16 = transpose(perm = var_3182, x = attn_output_45_cast_fp16)[name = string("transpose_57")]; + tensor var_3198_cast_fp16 = conv(dilations = var_3198_dilations_0, groups = var_3198_groups_0, pad = var_3198_pad_0, pad_type = var_3198_pad_type_0, strides = var_3198_strides_0, weight = squeeze_4_cast_fp16_to_fp32_to_fp16_palettized, x = var_3183_cast_fp16)[name = string("op_3198_cast_fp16")]; + tensor var_3202 = const()[name = string("op_3202"), val = tensor([0, 2, 1])]; + tensor attn_output_49_cast_fp16 = transpose(perm = var_3202, x = var_3198_cast_fp16)[name = string("transpose_56")]; + tensor hidden_states_49_cast_fp16 = add(x = hidden_states_41_cast_fp16, y = attn_output_49_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; + int32 var_3215 = const()[name = string("op_3215"), val = int32(-1)]; + fp16 const_146_promoted_to_fp16 = const()[name = string("const_146_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3217_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_146_promoted_to_fp16)[name = string("op_3217_cast_fp16")]; + bool input_83_interleave_0 = const()[name = string("input_83_interleave_0"), val = bool(false)]; + tensor input_83_cast_fp16 = concat(axis = var_3215, interleave = input_83_interleave_0, values = (hidden_states_49_cast_fp16, var_3217_cast_fp16))[name = string("input_83_cast_fp16")]; + tensor normed_77_axes_0 = const()[name = string("normed_77_axes_0"), val = tensor([-1])]; + fp16 var_3212_to_fp16 = const()[name = string("op_3212_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_77_cast_fp16 = layer_norm(axes = normed_77_axes_0, epsilon = var_3212_to_fp16, x = input_83_cast_fp16)[name = string("normed_77_cast_fp16")]; + tensor normed_79_begin_0 = const()[name = string("normed_79_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_79_end_0 = const()[name = string("normed_79_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_79_end_mask_0 = const()[name = string("normed_79_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_79_cast_fp16 = slice_by_index(begin = normed_79_begin_0, end = normed_79_end_0, end_mask = normed_79_end_mask_0, x = normed_77_cast_fp16)[name = string("normed_79_cast_fp16")]; + tensor const_149_promoted_to_fp16 = const()[name = string("const_149_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685165568)))]; + tensor x_77_cast_fp16 = mul(x = normed_79_cast_fp16, y = const_149_promoted_to_fp16)[name = string("x_77_cast_fp16")]; + tensor var_3242 = const()[name = string("op_3242"), val = tensor([0, 2, 1])]; + tensor input_85_axes_0 = const()[name = string("input_85_axes_0"), val = tensor([2])]; + tensor var_3243 = transpose(perm = var_3242, x = x_77_cast_fp16)[name = string("transpose_55")]; + tensor input_85 = expand_dims(axes = input_85_axes_0, x = var_3243)[name = string("input_85")]; + string input_87_pad_type_0 = const()[name = string("input_87_pad_type_0"), val = string("valid")]; + tensor input_87_strides_0 = const()[name = string("input_87_strides_0"), val = tensor([1, 1])]; + tensor input_87_pad_0 = const()[name = string("input_87_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_87_dilations_0 = const()[name = string("input_87_dilations_0"), val = tensor([1, 1])]; + int32 input_87_groups_0 = const()[name = string("input_87_groups_0"), val = int32(1)]; + tensor input_87 = conv(dilations = input_87_dilations_0, groups = input_87_groups_0, pad = input_87_pad_0, pad_type = input_87_pad_type_0, strides = input_87_strides_0, weight = model_model_layers_4_mlp_gate_proj_weight_palettized, x = input_85)[name = string("input_87")]; + string b_9_pad_type_0 = const()[name = string("b_9_pad_type_0"), val = string("valid")]; + tensor b_9_strides_0 = const()[name = string("b_9_strides_0"), val = tensor([1, 1])]; + tensor b_9_pad_0 = const()[name = string("b_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_9_dilations_0 = const()[name = string("b_9_dilations_0"), val = tensor([1, 1])]; + int32 b_9_groups_0 = const()[name = string("b_9_groups_0"), val = int32(1)]; + tensor b_9 = conv(dilations = b_9_dilations_0, groups = b_9_groups_0, pad = b_9_pad_0, pad_type = b_9_pad_type_0, strides = b_9_strides_0, weight = model_model_layers_4_mlp_up_proj_weight_palettized, x = input_85)[name = string("b_9")]; + tensor c_9 = silu(x = input_87)[name = string("c_9")]; + tensor input_89 = mul(x = c_9, y = b_9)[name = string("input_89")]; + string e_9_pad_type_0 = const()[name = string("e_9_pad_type_0"), val = string("valid")]; + tensor e_9_strides_0 = const()[name = string("e_9_strides_0"), val = tensor([1, 1])]; + tensor e_9_pad_0 = const()[name = string("e_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_9_dilations_0 = const()[name = string("e_9_dilations_0"), val = tensor([1, 1])]; + int32 e_9_groups_0 = const()[name = string("e_9_groups_0"), val = int32(1)]; + tensor e_9 = conv(dilations = e_9_dilations_0, groups = e_9_groups_0, pad = e_9_pad_0, pad_type = e_9_pad_type_0, strides = e_9_strides_0, weight = model_model_layers_4_mlp_down_proj_weight_palettized, x = input_89)[name = string("e_9")]; + tensor var_3265_axes_0 = const()[name = string("op_3265_axes_0"), val = tensor([2])]; + tensor var_3265 = squeeze(axes = var_3265_axes_0, x = e_9)[name = string("op_3265")]; + tensor var_3266 = const()[name = string("op_3266"), val = tensor([0, 2, 1])]; + tensor var_3267 = transpose(perm = var_3266, x = var_3265)[name = string("transpose_54")]; + tensor hidden_states_51_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = var_3267)[name = string("hidden_states_51_cast_fp16")]; + int32 var_3279 = const()[name = string("op_3279"), val = int32(-1)]; + fp16 const_150_promoted_to_fp16 = const()[name = string("const_150_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3281_cast_fp16 = mul(x = hidden_states_51_cast_fp16, y = const_150_promoted_to_fp16)[name = string("op_3281_cast_fp16")]; + bool input_91_interleave_0 = const()[name = string("input_91_interleave_0"), val = bool(false)]; + tensor input_91_cast_fp16 = concat(axis = var_3279, interleave = input_91_interleave_0, values = (hidden_states_51_cast_fp16, var_3281_cast_fp16))[name = string("input_91_cast_fp16")]; + tensor normed_81_axes_0 = const()[name = string("normed_81_axes_0"), val = tensor([-1])]; + fp16 var_3276_to_fp16 = const()[name = string("op_3276_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_81_cast_fp16 = layer_norm(axes = normed_81_axes_0, epsilon = var_3276_to_fp16, x = input_91_cast_fp16)[name = string("normed_81_cast_fp16")]; + tensor normed_83_begin_0 = const()[name = string("normed_83_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_83_end_0 = const()[name = string("normed_83_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_83_end_mask_0 = const()[name = string("normed_83_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_83_cast_fp16 = slice_by_index(begin = normed_83_begin_0, end = normed_83_end_0, end_mask = normed_83_end_mask_0, x = normed_81_cast_fp16)[name = string("normed_83_cast_fp16")]; + tensor const_153_promoted_to_fp16 = const()[name = string("const_153_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685169728)))]; + tensor hidden_states_53_cast_fp16 = mul(x = normed_83_cast_fp16, y = const_153_promoted_to_fp16)[name = string("hidden_states_53_cast_fp16")]; + tensor var_3298 = const()[name = string("op_3298"), val = tensor([0, 2, 1])]; + tensor var_3301_axes_0 = const()[name = string("op_3301_axes_0"), val = tensor([2])]; + tensor var_3299_cast_fp16 = transpose(perm = var_3298, x = hidden_states_53_cast_fp16)[name = string("transpose_53")]; + tensor var_3301_cast_fp16 = expand_dims(axes = var_3301_axes_0, x = var_3299_cast_fp16)[name = string("op_3301_cast_fp16")]; + string var_3317_pad_type_0 = const()[name = string("op_3317_pad_type_0"), val = string("valid")]; + tensor var_3317_strides_0 = const()[name = string("op_3317_strides_0"), val = tensor([1, 1])]; + tensor var_3317_pad_0 = const()[name = string("op_3317_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3317_dilations_0 = const()[name = string("op_3317_dilations_0"), val = tensor([1, 1])]; + int32 var_3317_groups_0 = const()[name = string("op_3317_groups_0"), val = int32(1)]; + tensor var_3317 = conv(dilations = var_3317_dilations_0, groups = var_3317_groups_0, pad = var_3317_pad_0, pad_type = var_3317_pad_type_0, strides = var_3317_strides_0, weight = model_model_layers_5_self_attn_q_proj_weight_palettized, x = var_3301_cast_fp16)[name = string("op_3317")]; + tensor var_3322 = const()[name = string("op_3322"), val = tensor([1, 16, 1, 128])]; + tensor var_3323 = reshape(shape = var_3322, x = var_3317)[name = string("op_3323")]; + string var_3339_pad_type_0 = const()[name = string("op_3339_pad_type_0"), val = string("valid")]; + tensor var_3339_strides_0 = const()[name = string("op_3339_strides_0"), val = tensor([1, 1])]; + tensor var_3339_pad_0 = const()[name = string("op_3339_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3339_dilations_0 = const()[name = string("op_3339_dilations_0"), val = tensor([1, 1])]; + int32 var_3339_groups_0 = const()[name = string("op_3339_groups_0"), val = int32(1)]; + tensor var_3339 = conv(dilations = var_3339_dilations_0, groups = var_3339_groups_0, pad = var_3339_pad_0, pad_type = var_3339_pad_type_0, strides = var_3339_strides_0, weight = model_model_layers_5_self_attn_k_proj_weight_palettized, x = var_3301_cast_fp16)[name = string("op_3339")]; + tensor var_3344 = const()[name = string("op_3344"), val = tensor([1, 8, 1, 128])]; + tensor var_3345 = reshape(shape = var_3344, x = var_3339)[name = string("op_3345")]; + string var_3361_pad_type_0 = const()[name = string("op_3361_pad_type_0"), val = string("valid")]; + tensor var_3361_strides_0 = const()[name = string("op_3361_strides_0"), val = tensor([1, 1])]; + tensor var_3361_pad_0 = const()[name = string("op_3361_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3361_dilations_0 = const()[name = string("op_3361_dilations_0"), val = tensor([1, 1])]; + int32 var_3361_groups_0 = const()[name = string("op_3361_groups_0"), val = int32(1)]; + tensor var_3361 = conv(dilations = var_3361_dilations_0, groups = var_3361_groups_0, pad = var_3361_pad_0, pad_type = var_3361_pad_type_0, strides = var_3361_strides_0, weight = model_model_layers_5_self_attn_v_proj_weight_palettized, x = var_3301_cast_fp16)[name = string("op_3361")]; + tensor var_3366 = const()[name = string("op_3366"), val = tensor([1, 8, 1, 128])]; + tensor var_3367 = reshape(shape = var_3366, x = var_3361)[name = string("op_3367")]; + int32 var_3382 = const()[name = string("op_3382"), val = int32(-1)]; + fp16 const_154_promoted = const()[name = string("const_154_promoted"), val = fp16(-0x1p+0)]; + tensor var_3384 = mul(x = var_3323, y = const_154_promoted)[name = string("op_3384")]; + bool input_95_interleave_0 = const()[name = string("input_95_interleave_0"), val = bool(false)]; + tensor input_95 = concat(axis = var_3382, interleave = input_95_interleave_0, values = (var_3323, var_3384))[name = string("input_95")]; + tensor normed_85_axes_0 = const()[name = string("normed_85_axes_0"), val = tensor([-1])]; + fp16 var_3379_to_fp16 = const()[name = string("op_3379_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_85_cast_fp16 = layer_norm(axes = normed_85_axes_0, epsilon = var_3379_to_fp16, x = input_95)[name = string("normed_85_cast_fp16")]; + tensor normed_87_begin_0 = const()[name = string("normed_87_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_87_end_0 = const()[name = string("normed_87_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_87_end_mask_0 = const()[name = string("normed_87_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_87 = slice_by_index(begin = normed_87_begin_0, end = normed_87_end_0, end_mask = normed_87_end_mask_0, x = normed_85_cast_fp16)[name = string("normed_87")]; + tensor const_157 = const()[name = string("const_157"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685173888)))]; + tensor q_11 = mul(x = normed_87, y = const_157)[name = string("q_11")]; + int32 var_3407 = const()[name = string("op_3407"), val = int32(-1)]; + fp16 const_158_promoted = const()[name = string("const_158_promoted"), val = fp16(-0x1p+0)]; + tensor var_3409 = mul(x = var_3345, y = const_158_promoted)[name = string("op_3409")]; + bool input_97_interleave_0 = const()[name = string("input_97_interleave_0"), val = bool(false)]; + tensor input_97 = concat(axis = var_3407, interleave = input_97_interleave_0, values = (var_3345, var_3409))[name = string("input_97")]; + tensor normed_89_axes_0 = const()[name = string("normed_89_axes_0"), val = tensor([-1])]; + fp16 var_3404_to_fp16 = const()[name = string("op_3404_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_89_cast_fp16 = layer_norm(axes = normed_89_axes_0, epsilon = var_3404_to_fp16, x = input_97)[name = string("normed_89_cast_fp16")]; + tensor normed_91_begin_0 = const()[name = string("normed_91_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_91_end_0 = const()[name = string("normed_91_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_91_end_mask_0 = const()[name = string("normed_91_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_91 = slice_by_index(begin = normed_91_begin_0, end = normed_91_end_0, end_mask = normed_91_end_mask_0, x = normed_89_cast_fp16)[name = string("normed_91")]; + tensor const_161 = const()[name = string("const_161"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685174208)))]; + tensor k_11 = mul(x = normed_91, y = const_161)[name = string("k_11")]; + tensor var_3423 = mul(x = q_11, y = cos_1_cast_fp16)[name = string("op_3423")]; + tensor x1_21_begin_0 = const()[name = string("x1_21_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_21_end_0 = const()[name = string("x1_21_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_21_end_mask_0 = const()[name = string("x1_21_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_21 = slice_by_index(begin = x1_21_begin_0, end = x1_21_end_0, end_mask = x1_21_end_mask_0, x = q_11)[name = string("x1_21")]; + tensor x2_21_begin_0 = const()[name = string("x2_21_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_21_end_0 = const()[name = string("x2_21_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_21_end_mask_0 = const()[name = string("x2_21_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_21 = slice_by_index(begin = x2_21_begin_0, end = x2_21_end_0, end_mask = x2_21_end_mask_0, x = q_11)[name = string("x2_21")]; + fp16 const_164_promoted = const()[name = string("const_164_promoted"), val = fp16(-0x1p+0)]; + tensor var_3444 = mul(x = x2_21, y = const_164_promoted)[name = string("op_3444")]; + int32 var_3446 = const()[name = string("op_3446"), val = int32(-1)]; + bool var_3447_interleave_0 = const()[name = string("op_3447_interleave_0"), val = bool(false)]; + tensor var_3447 = concat(axis = var_3446, interleave = var_3447_interleave_0, values = (var_3444, x1_21))[name = string("op_3447")]; + tensor var_3448 = mul(x = var_3447, y = sin_1_cast_fp16)[name = string("op_3448")]; + tensor query_states_21 = add(x = var_3423, y = var_3448)[name = string("query_states_21")]; + tensor var_3451 = mul(x = k_11, y = cos_1_cast_fp16)[name = string("op_3451")]; + tensor x1_23_begin_0 = const()[name = string("x1_23_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_23_end_0 = const()[name = string("x1_23_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_23_end_mask_0 = const()[name = string("x1_23_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_23 = slice_by_index(begin = x1_23_begin_0, end = x1_23_end_0, end_mask = x1_23_end_mask_0, x = k_11)[name = string("x1_23")]; + tensor x2_23_begin_0 = const()[name = string("x2_23_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_23_end_0 = const()[name = string("x2_23_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_23_end_mask_0 = const()[name = string("x2_23_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_23 = slice_by_index(begin = x2_23_begin_0, end = x2_23_end_0, end_mask = x2_23_end_mask_0, x = k_11)[name = string("x2_23")]; + fp16 const_167_promoted = const()[name = string("const_167_promoted"), val = fp16(-0x1p+0)]; + tensor var_3472 = mul(x = x2_23, y = const_167_promoted)[name = string("op_3472")]; + int32 var_3474 = const()[name = string("op_3474"), val = int32(-1)]; + bool var_3475_interleave_0 = const()[name = string("op_3475_interleave_0"), val = bool(false)]; + tensor var_3475 = concat(axis = var_3474, interleave = var_3475_interleave_0, values = (var_3472, x1_23))[name = string("op_3475")]; + tensor var_3476 = mul(x = var_3475, y = sin_1_cast_fp16)[name = string("op_3476")]; + tensor key_states_21 = add(x = var_3451, y = var_3476)[name = string("key_states_21")]; + tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([5])]; + tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; + tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; + tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([6])]; + int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; + bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; + tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_60, expand_dims_61, current_pos, expand_dims_63))[name = string("concat_42")]; + tensor concat_43_values1_0 = const()[name = string("concat_43_values1_0"), val = tensor([0])]; + tensor concat_43_values3_0 = const()[name = string("concat_43_values3_0"), val = tensor([0])]; + int32 concat_43_axis_0 = const()[name = string("concat_43_axis_0"), val = int32(0)]; + bool concat_43_interleave_0 = const()[name = string("concat_43_interleave_0"), val = bool(false)]; + tensor concat_43 = concat(axis = concat_43_axis_0, interleave = concat_43_interleave_0, values = (expand_dims_64, concat_43_values1_0, var_1001, concat_43_values3_0))[name = string("concat_43")]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_42, begin_mask = model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0, end = concat_43, end_mask = model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_11_stride_0, update = key_states_21, x = coreml_update_state_37)[name = string("model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_66_write_state")]; + tensor coreml_update_state_38 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_66")]; + tensor expand_dims_66 = const()[name = string("expand_dims_66"), val = tensor([33])]; + tensor expand_dims_67 = const()[name = string("expand_dims_67"), val = tensor([0])]; + tensor expand_dims_69 = const()[name = string("expand_dims_69"), val = tensor([0])]; + tensor expand_dims_70 = const()[name = string("expand_dims_70"), val = tensor([34])]; + int32 concat_46_axis_0 = const()[name = string("concat_46_axis_0"), val = int32(0)]; + bool concat_46_interleave_0 = const()[name = string("concat_46_interleave_0"), val = bool(false)]; + tensor concat_46 = concat(axis = concat_46_axis_0, interleave = concat_46_interleave_0, values = (expand_dims_66, expand_dims_67, current_pos, expand_dims_69))[name = string("concat_46")]; + tensor concat_47_values1_0 = const()[name = string("concat_47_values1_0"), val = tensor([0])]; + tensor concat_47_values3_0 = const()[name = string("concat_47_values3_0"), val = tensor([0])]; + int32 concat_47_axis_0 = const()[name = string("concat_47_axis_0"), val = int32(0)]; + bool concat_47_interleave_0 = const()[name = string("concat_47_interleave_0"), val = bool(false)]; + tensor concat_47 = concat(axis = concat_47_axis_0, interleave = concat_47_interleave_0, values = (expand_dims_70, concat_47_values1_0, var_1001, concat_47_values3_0))[name = string("concat_47")]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_46, begin_mask = model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0, end = concat_47, end_mask = model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_12_stride_0, update = var_3367, x = coreml_update_state_38)[name = string("model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_67_write_state")]; + tensor coreml_update_state_39 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_67")]; + tensor var_3531_begin_0 = const()[name = string("op_3531_begin_0"), val = tensor([5, 0, 0, 0])]; + tensor var_3531_end_0 = const()[name = string("op_3531_end_0"), val = tensor([6, 8, 1024, 128])]; + tensor var_3531_end_mask_0 = const()[name = string("op_3531_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3531_cast_fp16 = slice_by_index(begin = var_3531_begin_0, end = var_3531_end_0, end_mask = var_3531_end_mask_0, x = coreml_update_state_39)[name = string("op_3531_cast_fp16")]; + tensor K_layer_cache_11_axes_0 = const()[name = string("K_layer_cache_11_axes_0"), val = tensor([0])]; + tensor K_layer_cache_11_cast_fp16 = squeeze(axes = K_layer_cache_11_axes_0, x = var_3531_cast_fp16)[name = string("K_layer_cache_11_cast_fp16")]; + tensor var_3538_begin_0 = const()[name = string("op_3538_begin_0"), val = tensor([33, 0, 0, 0])]; + tensor var_3538_end_0 = const()[name = string("op_3538_end_0"), val = tensor([34, 8, 1024, 128])]; + tensor var_3538_end_mask_0 = const()[name = string("op_3538_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3538_cast_fp16 = slice_by_index(begin = var_3538_begin_0, end = var_3538_end_0, end_mask = var_3538_end_mask_0, x = coreml_update_state_39)[name = string("op_3538_cast_fp16")]; + tensor V_layer_cache_11_axes_0 = const()[name = string("V_layer_cache_11_axes_0"), val = tensor([0])]; + tensor V_layer_cache_11_cast_fp16 = squeeze(axes = V_layer_cache_11_axes_0, x = var_3538_cast_fp16)[name = string("V_layer_cache_11_cast_fp16")]; + tensor x_83_axes_0 = const()[name = string("x_83_axes_0"), val = tensor([1])]; + tensor x_83_cast_fp16 = expand_dims(axes = x_83_axes_0, x = K_layer_cache_11_cast_fp16)[name = string("x_83_cast_fp16")]; + tensor var_3575 = const()[name = string("op_3575"), val = tensor([1, 2, 1, 1])]; + tensor x_85_cast_fp16 = tile(reps = var_3575, x = x_83_cast_fp16)[name = string("x_85_cast_fp16")]; + tensor var_3587 = const()[name = string("op_3587"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_23_cast_fp16 = reshape(shape = var_3587, x = x_85_cast_fp16)[name = string("key_states_23_cast_fp16")]; + tensor x_89_axes_0 = const()[name = string("x_89_axes_0"), val = tensor([1])]; + tensor x_89_cast_fp16 = expand_dims(axes = x_89_axes_0, x = V_layer_cache_11_cast_fp16)[name = string("x_89_cast_fp16")]; + tensor var_3595 = const()[name = string("op_3595"), val = tensor([1, 2, 1, 1])]; + tensor x_91_cast_fp16 = tile(reps = var_3595, x = x_89_cast_fp16)[name = string("x_91_cast_fp16")]; + tensor var_3607 = const()[name = string("op_3607"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_33_cast_fp16 = reshape(shape = var_3607, x = x_91_cast_fp16)[name = string("value_states_33_cast_fp16")]; + bool var_3622_transpose_x_1 = const()[name = string("op_3622_transpose_x_1"), val = bool(false)]; + bool var_3622_transpose_y_1 = const()[name = string("op_3622_transpose_y_1"), val = bool(true)]; + tensor var_3622 = matmul(transpose_x = var_3622_transpose_x_1, transpose_y = var_3622_transpose_y_1, x = query_states_21, y = key_states_23_cast_fp16)[name = string("op_3622")]; + fp16 var_3623_to_fp16 = const()[name = string("op_3623_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_31_cast_fp16 = mul(x = var_3622, y = var_3623_to_fp16)[name = string("attn_weights_31_cast_fp16")]; + tensor attn_weights_33_cast_fp16 = add(x = attn_weights_31_cast_fp16, y = causal_mask)[name = string("attn_weights_33_cast_fp16")]; + int32 var_3658 = const()[name = string("op_3658"), val = int32(-1)]; + tensor attn_weights_35_cast_fp16 = softmax(axis = var_3658, x = attn_weights_33_cast_fp16)[name = string("attn_weights_35_cast_fp16")]; + bool attn_output_51_transpose_x_0 = const()[name = string("attn_output_51_transpose_x_0"), val = bool(false)]; + bool attn_output_51_transpose_y_0 = const()[name = string("attn_output_51_transpose_y_0"), val = bool(false)]; + tensor attn_output_51_cast_fp16 = matmul(transpose_x = attn_output_51_transpose_x_0, transpose_y = attn_output_51_transpose_y_0, x = attn_weights_35_cast_fp16, y = value_states_33_cast_fp16)[name = string("attn_output_51_cast_fp16")]; + tensor var_3669_perm_0 = const()[name = string("op_3669_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_3673 = const()[name = string("op_3673"), val = tensor([1, 1, 2048])]; + tensor var_3669_cast_fp16 = transpose(perm = var_3669_perm_0, x = attn_output_51_cast_fp16)[name = string("transpose_52")]; + tensor attn_output_55_cast_fp16 = reshape(shape = var_3673, x = var_3669_cast_fp16)[name = string("attn_output_55_cast_fp16")]; + tensor var_3678 = const()[name = string("op_3678"), val = tensor([0, 2, 1])]; + string var_3694_pad_type_0 = const()[name = string("op_3694_pad_type_0"), val = string("valid")]; + int32 var_3694_groups_0 = const()[name = string("op_3694_groups_0"), val = int32(1)]; + tensor var_3694_strides_0 = const()[name = string("op_3694_strides_0"), val = tensor([1])]; + tensor var_3694_pad_0 = const()[name = string("op_3694_pad_0"), val = tensor([0, 0])]; + tensor var_3694_dilations_0 = const()[name = string("op_3694_dilations_0"), val = tensor([1])]; + tensor squeeze_5_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685174528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689368896))))[name = string("squeeze_5_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_3679_cast_fp16 = transpose(perm = var_3678, x = attn_output_55_cast_fp16)[name = string("transpose_51")]; + tensor var_3694_cast_fp16 = conv(dilations = var_3694_dilations_0, groups = var_3694_groups_0, pad = var_3694_pad_0, pad_type = var_3694_pad_type_0, strides = var_3694_strides_0, weight = squeeze_5_cast_fp16_to_fp32_to_fp16_palettized, x = var_3679_cast_fp16)[name = string("op_3694_cast_fp16")]; + tensor var_3698 = const()[name = string("op_3698"), val = tensor([0, 2, 1])]; + tensor attn_output_59_cast_fp16 = transpose(perm = var_3698, x = var_3694_cast_fp16)[name = string("transpose_50")]; + tensor hidden_states_59_cast_fp16 = add(x = hidden_states_51_cast_fp16, y = attn_output_59_cast_fp16)[name = string("hidden_states_59_cast_fp16")]; + int32 var_3711 = const()[name = string("op_3711"), val = int32(-1)]; + fp16 const_176_promoted_to_fp16 = const()[name = string("const_176_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3713_cast_fp16 = mul(x = hidden_states_59_cast_fp16, y = const_176_promoted_to_fp16)[name = string("op_3713_cast_fp16")]; + bool input_101_interleave_0 = const()[name = string("input_101_interleave_0"), val = bool(false)]; + tensor input_101_cast_fp16 = concat(axis = var_3711, interleave = input_101_interleave_0, values = (hidden_states_59_cast_fp16, var_3713_cast_fp16))[name = string("input_101_cast_fp16")]; + tensor normed_93_axes_0 = const()[name = string("normed_93_axes_0"), val = tensor([-1])]; + fp16 var_3708_to_fp16 = const()[name = string("op_3708_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_93_cast_fp16 = layer_norm(axes = normed_93_axes_0, epsilon = var_3708_to_fp16, x = input_101_cast_fp16)[name = string("normed_93_cast_fp16")]; + tensor normed_95_begin_0 = const()[name = string("normed_95_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_95_end_0 = const()[name = string("normed_95_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_95_end_mask_0 = const()[name = string("normed_95_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_95_cast_fp16 = slice_by_index(begin = normed_95_begin_0, end = normed_95_end_0, end_mask = normed_95_end_mask_0, x = normed_93_cast_fp16)[name = string("normed_95_cast_fp16")]; + tensor const_179_promoted_to_fp16 = const()[name = string("const_179_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689500032)))]; + tensor x_93_cast_fp16 = mul(x = normed_95_cast_fp16, y = const_179_promoted_to_fp16)[name = string("x_93_cast_fp16")]; + tensor var_3738 = const()[name = string("op_3738"), val = tensor([0, 2, 1])]; + tensor input_103_axes_0 = const()[name = string("input_103_axes_0"), val = tensor([2])]; + tensor var_3739 = transpose(perm = var_3738, x = x_93_cast_fp16)[name = string("transpose_49")]; + tensor input_103 = expand_dims(axes = input_103_axes_0, x = var_3739)[name = string("input_103")]; + string input_105_pad_type_0 = const()[name = string("input_105_pad_type_0"), val = string("valid")]; + tensor input_105_strides_0 = const()[name = string("input_105_strides_0"), val = tensor([1, 1])]; + tensor input_105_pad_0 = const()[name = string("input_105_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_105_dilations_0 = const()[name = string("input_105_dilations_0"), val = tensor([1, 1])]; + int32 input_105_groups_0 = const()[name = string("input_105_groups_0"), val = int32(1)]; + tensor input_105 = conv(dilations = input_105_dilations_0, groups = input_105_groups_0, pad = input_105_pad_0, pad_type = input_105_pad_type_0, strides = input_105_strides_0, weight = model_model_layers_5_mlp_gate_proj_weight_palettized, x = input_103)[name = string("input_105")]; + string b_11_pad_type_0 = const()[name = string("b_11_pad_type_0"), val = string("valid")]; + tensor b_11_strides_0 = const()[name = string("b_11_strides_0"), val = tensor([1, 1])]; + tensor b_11_pad_0 = const()[name = string("b_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_11_dilations_0 = const()[name = string("b_11_dilations_0"), val = tensor([1, 1])]; + int32 b_11_groups_0 = const()[name = string("b_11_groups_0"), val = int32(1)]; + tensor b_11 = conv(dilations = b_11_dilations_0, groups = b_11_groups_0, pad = b_11_pad_0, pad_type = b_11_pad_type_0, strides = b_11_strides_0, weight = model_model_layers_5_mlp_up_proj_weight_palettized, x = input_103)[name = string("b_11")]; + tensor c_11 = silu(x = input_105)[name = string("c_11")]; + tensor input_107 = mul(x = c_11, y = b_11)[name = string("input_107")]; + string e_11_pad_type_0 = const()[name = string("e_11_pad_type_0"), val = string("valid")]; + tensor e_11_strides_0 = const()[name = string("e_11_strides_0"), val = tensor([1, 1])]; + tensor e_11_pad_0 = const()[name = string("e_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_11_dilations_0 = const()[name = string("e_11_dilations_0"), val = tensor([1, 1])]; + int32 e_11_groups_0 = const()[name = string("e_11_groups_0"), val = int32(1)]; + tensor e_11 = conv(dilations = e_11_dilations_0, groups = e_11_groups_0, pad = e_11_pad_0, pad_type = e_11_pad_type_0, strides = e_11_strides_0, weight = model_model_layers_5_mlp_down_proj_weight_palettized, x = input_107)[name = string("e_11")]; + tensor var_3761_axes_0 = const()[name = string("op_3761_axes_0"), val = tensor([2])]; + tensor var_3761 = squeeze(axes = var_3761_axes_0, x = e_11)[name = string("op_3761")]; + tensor var_3762 = const()[name = string("op_3762"), val = tensor([0, 2, 1])]; + tensor var_3763 = transpose(perm = var_3762, x = var_3761)[name = string("transpose_48")]; + tensor hidden_states_61_cast_fp16 = add(x = hidden_states_59_cast_fp16, y = var_3763)[name = string("hidden_states_61_cast_fp16")]; + int32 var_3775 = const()[name = string("op_3775"), val = int32(-1)]; + fp16 const_180_promoted_to_fp16 = const()[name = string("const_180_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3777_cast_fp16 = mul(x = hidden_states_61_cast_fp16, y = const_180_promoted_to_fp16)[name = string("op_3777_cast_fp16")]; + bool input_109_interleave_0 = const()[name = string("input_109_interleave_0"), val = bool(false)]; + tensor input_109_cast_fp16 = concat(axis = var_3775, interleave = input_109_interleave_0, values = (hidden_states_61_cast_fp16, var_3777_cast_fp16))[name = string("input_109_cast_fp16")]; + tensor normed_97_axes_0 = const()[name = string("normed_97_axes_0"), val = tensor([-1])]; + fp16 var_3772_to_fp16 = const()[name = string("op_3772_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_97_cast_fp16 = layer_norm(axes = normed_97_axes_0, epsilon = var_3772_to_fp16, x = input_109_cast_fp16)[name = string("normed_97_cast_fp16")]; + tensor normed_99_begin_0 = const()[name = string("normed_99_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_99_end_0 = const()[name = string("normed_99_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_99_end_mask_0 = const()[name = string("normed_99_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_99_cast_fp16 = slice_by_index(begin = normed_99_begin_0, end = normed_99_end_0, end_mask = normed_99_end_mask_0, x = normed_97_cast_fp16)[name = string("normed_99_cast_fp16")]; + tensor const_183_promoted_to_fp16 = const()[name = string("const_183_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689504192)))]; + tensor hidden_states_63_cast_fp16 = mul(x = normed_99_cast_fp16, y = const_183_promoted_to_fp16)[name = string("hidden_states_63_cast_fp16")]; + tensor var_3794 = const()[name = string("op_3794"), val = tensor([0, 2, 1])]; + tensor var_3797_axes_0 = const()[name = string("op_3797_axes_0"), val = tensor([2])]; + tensor var_3795_cast_fp16 = transpose(perm = var_3794, x = hidden_states_63_cast_fp16)[name = string("transpose_47")]; + tensor var_3797_cast_fp16 = expand_dims(axes = var_3797_axes_0, x = var_3795_cast_fp16)[name = string("op_3797_cast_fp16")]; + string var_3813_pad_type_0 = const()[name = string("op_3813_pad_type_0"), val = string("valid")]; + tensor var_3813_strides_0 = const()[name = string("op_3813_strides_0"), val = tensor([1, 1])]; + tensor var_3813_pad_0 = const()[name = string("op_3813_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3813_dilations_0 = const()[name = string("op_3813_dilations_0"), val = tensor([1, 1])]; + int32 var_3813_groups_0 = const()[name = string("op_3813_groups_0"), val = int32(1)]; + tensor var_3813 = conv(dilations = var_3813_dilations_0, groups = var_3813_groups_0, pad = var_3813_pad_0, pad_type = var_3813_pad_type_0, strides = var_3813_strides_0, weight = model_model_layers_6_self_attn_q_proj_weight_palettized, x = var_3797_cast_fp16)[name = string("op_3813")]; + tensor var_3818 = const()[name = string("op_3818"), val = tensor([1, 16, 1, 128])]; + tensor var_3819 = reshape(shape = var_3818, x = var_3813)[name = string("op_3819")]; + string var_3835_pad_type_0 = const()[name = string("op_3835_pad_type_0"), val = string("valid")]; + tensor var_3835_strides_0 = const()[name = string("op_3835_strides_0"), val = tensor([1, 1])]; + tensor var_3835_pad_0 = const()[name = string("op_3835_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3835_dilations_0 = const()[name = string("op_3835_dilations_0"), val = tensor([1, 1])]; + int32 var_3835_groups_0 = const()[name = string("op_3835_groups_0"), val = int32(1)]; + tensor var_3835 = conv(dilations = var_3835_dilations_0, groups = var_3835_groups_0, pad = var_3835_pad_0, pad_type = var_3835_pad_type_0, strides = var_3835_strides_0, weight = model_model_layers_6_self_attn_k_proj_weight_palettized, x = var_3797_cast_fp16)[name = string("op_3835")]; + tensor var_3840 = const()[name = string("op_3840"), val = tensor([1, 8, 1, 128])]; + tensor var_3841 = reshape(shape = var_3840, x = var_3835)[name = string("op_3841")]; + string var_3857_pad_type_0 = const()[name = string("op_3857_pad_type_0"), val = string("valid")]; + tensor var_3857_strides_0 = const()[name = string("op_3857_strides_0"), val = tensor([1, 1])]; + tensor var_3857_pad_0 = const()[name = string("op_3857_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3857_dilations_0 = const()[name = string("op_3857_dilations_0"), val = tensor([1, 1])]; + int32 var_3857_groups_0 = const()[name = string("op_3857_groups_0"), val = int32(1)]; + tensor var_3857 = conv(dilations = var_3857_dilations_0, groups = var_3857_groups_0, pad = var_3857_pad_0, pad_type = var_3857_pad_type_0, strides = var_3857_strides_0, weight = model_model_layers_6_self_attn_v_proj_weight_palettized, x = var_3797_cast_fp16)[name = string("op_3857")]; + tensor var_3862 = const()[name = string("op_3862"), val = tensor([1, 8, 1, 128])]; + tensor var_3863 = reshape(shape = var_3862, x = var_3857)[name = string("op_3863")]; + int32 var_3878 = const()[name = string("op_3878"), val = int32(-1)]; + fp16 const_184_promoted = const()[name = string("const_184_promoted"), val = fp16(-0x1p+0)]; + tensor var_3880 = mul(x = var_3819, y = const_184_promoted)[name = string("op_3880")]; + bool input_113_interleave_0 = const()[name = string("input_113_interleave_0"), val = bool(false)]; + tensor input_113 = concat(axis = var_3878, interleave = input_113_interleave_0, values = (var_3819, var_3880))[name = string("input_113")]; + tensor normed_101_axes_0 = const()[name = string("normed_101_axes_0"), val = tensor([-1])]; + fp16 var_3875_to_fp16 = const()[name = string("op_3875_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_101_cast_fp16 = layer_norm(axes = normed_101_axes_0, epsilon = var_3875_to_fp16, x = input_113)[name = string("normed_101_cast_fp16")]; + tensor normed_103_begin_0 = const()[name = string("normed_103_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_103_end_0 = const()[name = string("normed_103_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_103_end_mask_0 = const()[name = string("normed_103_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_103 = slice_by_index(begin = normed_103_begin_0, end = normed_103_end_0, end_mask = normed_103_end_mask_0, x = normed_101_cast_fp16)[name = string("normed_103")]; + tensor const_187 = const()[name = string("const_187"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689508352)))]; + tensor q_13 = mul(x = normed_103, y = const_187)[name = string("q_13")]; + int32 var_3903 = const()[name = string("op_3903"), val = int32(-1)]; + fp16 const_188_promoted = const()[name = string("const_188_promoted"), val = fp16(-0x1p+0)]; + tensor var_3905 = mul(x = var_3841, y = const_188_promoted)[name = string("op_3905")]; + bool input_115_interleave_0 = const()[name = string("input_115_interleave_0"), val = bool(false)]; + tensor input_115 = concat(axis = var_3903, interleave = input_115_interleave_0, values = (var_3841, var_3905))[name = string("input_115")]; + tensor normed_105_axes_0 = const()[name = string("normed_105_axes_0"), val = tensor([-1])]; + fp16 var_3900_to_fp16 = const()[name = string("op_3900_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_105_cast_fp16 = layer_norm(axes = normed_105_axes_0, epsilon = var_3900_to_fp16, x = input_115)[name = string("normed_105_cast_fp16")]; + tensor normed_107_begin_0 = const()[name = string("normed_107_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_107_end_0 = const()[name = string("normed_107_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_107_end_mask_0 = const()[name = string("normed_107_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_107 = slice_by_index(begin = normed_107_begin_0, end = normed_107_end_0, end_mask = normed_107_end_mask_0, x = normed_105_cast_fp16)[name = string("normed_107")]; + tensor const_191 = const()[name = string("const_191"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689508672)))]; + tensor k_13 = mul(x = normed_107, y = const_191)[name = string("k_13")]; + tensor var_3919 = mul(x = q_13, y = cos_1_cast_fp16)[name = string("op_3919")]; + tensor x1_25_begin_0 = const()[name = string("x1_25_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_25_end_0 = const()[name = string("x1_25_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_25_end_mask_0 = const()[name = string("x1_25_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_25 = slice_by_index(begin = x1_25_begin_0, end = x1_25_end_0, end_mask = x1_25_end_mask_0, x = q_13)[name = string("x1_25")]; + tensor x2_25_begin_0 = const()[name = string("x2_25_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_25_end_0 = const()[name = string("x2_25_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_25_end_mask_0 = const()[name = string("x2_25_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_25 = slice_by_index(begin = x2_25_begin_0, end = x2_25_end_0, end_mask = x2_25_end_mask_0, x = q_13)[name = string("x2_25")]; + fp16 const_194_promoted = const()[name = string("const_194_promoted"), val = fp16(-0x1p+0)]; + tensor var_3940 = mul(x = x2_25, y = const_194_promoted)[name = string("op_3940")]; + int32 var_3942 = const()[name = string("op_3942"), val = int32(-1)]; + bool var_3943_interleave_0 = const()[name = string("op_3943_interleave_0"), val = bool(false)]; + tensor var_3943 = concat(axis = var_3942, interleave = var_3943_interleave_0, values = (var_3940, x1_25))[name = string("op_3943")]; + tensor var_3944 = mul(x = var_3943, y = sin_1_cast_fp16)[name = string("op_3944")]; + tensor query_states_25 = add(x = var_3919, y = var_3944)[name = string("query_states_25")]; + tensor var_3947 = mul(x = k_13, y = cos_1_cast_fp16)[name = string("op_3947")]; + tensor x1_27_begin_0 = const()[name = string("x1_27_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_27_end_0 = const()[name = string("x1_27_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_27_end_mask_0 = const()[name = string("x1_27_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_27 = slice_by_index(begin = x1_27_begin_0, end = x1_27_end_0, end_mask = x1_27_end_mask_0, x = k_13)[name = string("x1_27")]; + tensor x2_27_begin_0 = const()[name = string("x2_27_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_27_end_0 = const()[name = string("x2_27_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_27_end_mask_0 = const()[name = string("x2_27_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_27 = slice_by_index(begin = x2_27_begin_0, end = x2_27_end_0, end_mask = x2_27_end_mask_0, x = k_13)[name = string("x2_27")]; + fp16 const_197_promoted = const()[name = string("const_197_promoted"), val = fp16(-0x1p+0)]; + tensor var_3968 = mul(x = x2_27, y = const_197_promoted)[name = string("op_3968")]; + int32 var_3970 = const()[name = string("op_3970"), val = int32(-1)]; + bool var_3971_interleave_0 = const()[name = string("op_3971_interleave_0"), val = bool(false)]; + tensor var_3971 = concat(axis = var_3970, interleave = var_3971_interleave_0, values = (var_3968, x1_27))[name = string("op_3971")]; + tensor var_3972 = mul(x = var_3971, y = sin_1_cast_fp16)[name = string("op_3972")]; + tensor key_states_25 = add(x = var_3947, y = var_3972)[name = string("key_states_25")]; + tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([6])]; + tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; + tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; + tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([7])]; + int32 concat_50_axis_0 = const()[name = string("concat_50_axis_0"), val = int32(0)]; + bool concat_50_interleave_0 = const()[name = string("concat_50_interleave_0"), val = bool(false)]; + tensor concat_50 = concat(axis = concat_50_axis_0, interleave = concat_50_interleave_0, values = (expand_dims_72, expand_dims_73, current_pos, expand_dims_75))[name = string("concat_50")]; + tensor concat_51_values1_0 = const()[name = string("concat_51_values1_0"), val = tensor([0])]; + tensor concat_51_values3_0 = const()[name = string("concat_51_values3_0"), val = tensor([0])]; + int32 concat_51_axis_0 = const()[name = string("concat_51_axis_0"), val = int32(0)]; + bool concat_51_interleave_0 = const()[name = string("concat_51_interleave_0"), val = bool(false)]; + tensor concat_51 = concat(axis = concat_51_axis_0, interleave = concat_51_interleave_0, values = (expand_dims_76, concat_51_values1_0, var_1001, concat_51_values3_0))[name = string("concat_51")]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_50, begin_mask = model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0, end = concat_51, end_mask = model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_13_stride_0, update = key_states_25, x = coreml_update_state_39)[name = string("model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_68_write_state")]; + tensor coreml_update_state_40 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_68")]; + tensor expand_dims_78 = const()[name = string("expand_dims_78"), val = tensor([34])]; + tensor expand_dims_79 = const()[name = string("expand_dims_79"), val = tensor([0])]; + tensor expand_dims_81 = const()[name = string("expand_dims_81"), val = tensor([0])]; + tensor expand_dims_82 = const()[name = string("expand_dims_82"), val = tensor([35])]; + int32 concat_54_axis_0 = const()[name = string("concat_54_axis_0"), val = int32(0)]; + bool concat_54_interleave_0 = const()[name = string("concat_54_interleave_0"), val = bool(false)]; + tensor concat_54 = concat(axis = concat_54_axis_0, interleave = concat_54_interleave_0, values = (expand_dims_78, expand_dims_79, current_pos, expand_dims_81))[name = string("concat_54")]; + tensor concat_55_values1_0 = const()[name = string("concat_55_values1_0"), val = tensor([0])]; + tensor concat_55_values3_0 = const()[name = string("concat_55_values3_0"), val = tensor([0])]; + int32 concat_55_axis_0 = const()[name = string("concat_55_axis_0"), val = int32(0)]; + bool concat_55_interleave_0 = const()[name = string("concat_55_interleave_0"), val = bool(false)]; + tensor concat_55 = concat(axis = concat_55_axis_0, interleave = concat_55_interleave_0, values = (expand_dims_82, concat_55_values1_0, var_1001, concat_55_values3_0))[name = string("concat_55")]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_54, begin_mask = model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0, end = concat_55, end_mask = model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_14_stride_0, update = var_3863, x = coreml_update_state_40)[name = string("model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_69_write_state")]; + tensor coreml_update_state_41 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_69")]; + tensor var_4027_begin_0 = const()[name = string("op_4027_begin_0"), val = tensor([6, 0, 0, 0])]; + tensor var_4027_end_0 = const()[name = string("op_4027_end_0"), val = tensor([7, 8, 1024, 128])]; + tensor var_4027_end_mask_0 = const()[name = string("op_4027_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4027_cast_fp16 = slice_by_index(begin = var_4027_begin_0, end = var_4027_end_0, end_mask = var_4027_end_mask_0, x = coreml_update_state_41)[name = string("op_4027_cast_fp16")]; + tensor K_layer_cache_13_axes_0 = const()[name = string("K_layer_cache_13_axes_0"), val = tensor([0])]; + tensor K_layer_cache_13_cast_fp16 = squeeze(axes = K_layer_cache_13_axes_0, x = var_4027_cast_fp16)[name = string("K_layer_cache_13_cast_fp16")]; + tensor var_4034_begin_0 = const()[name = string("op_4034_begin_0"), val = tensor([34, 0, 0, 0])]; + tensor var_4034_end_0 = const()[name = string("op_4034_end_0"), val = tensor([35, 8, 1024, 128])]; + tensor var_4034_end_mask_0 = const()[name = string("op_4034_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4034_cast_fp16 = slice_by_index(begin = var_4034_begin_0, end = var_4034_end_0, end_mask = var_4034_end_mask_0, x = coreml_update_state_41)[name = string("op_4034_cast_fp16")]; + tensor V_layer_cache_13_axes_0 = const()[name = string("V_layer_cache_13_axes_0"), val = tensor([0])]; + tensor V_layer_cache_13_cast_fp16 = squeeze(axes = V_layer_cache_13_axes_0, x = var_4034_cast_fp16)[name = string("V_layer_cache_13_cast_fp16")]; + tensor x_99_axes_0 = const()[name = string("x_99_axes_0"), val = tensor([1])]; + tensor x_99_cast_fp16 = expand_dims(axes = x_99_axes_0, x = K_layer_cache_13_cast_fp16)[name = string("x_99_cast_fp16")]; + tensor var_4071 = const()[name = string("op_4071"), val = tensor([1, 2, 1, 1])]; + tensor x_101_cast_fp16 = tile(reps = var_4071, x = x_99_cast_fp16)[name = string("x_101_cast_fp16")]; + tensor var_4083 = const()[name = string("op_4083"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_27_cast_fp16 = reshape(shape = var_4083, x = x_101_cast_fp16)[name = string("key_states_27_cast_fp16")]; + tensor x_105_axes_0 = const()[name = string("x_105_axes_0"), val = tensor([1])]; + tensor x_105_cast_fp16 = expand_dims(axes = x_105_axes_0, x = V_layer_cache_13_cast_fp16)[name = string("x_105_cast_fp16")]; + tensor var_4091 = const()[name = string("op_4091"), val = tensor([1, 2, 1, 1])]; + tensor x_107_cast_fp16 = tile(reps = var_4091, x = x_105_cast_fp16)[name = string("x_107_cast_fp16")]; + tensor var_4103 = const()[name = string("op_4103"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_39_cast_fp16 = reshape(shape = var_4103, x = x_107_cast_fp16)[name = string("value_states_39_cast_fp16")]; + bool var_4118_transpose_x_1 = const()[name = string("op_4118_transpose_x_1"), val = bool(false)]; + bool var_4118_transpose_y_1 = const()[name = string("op_4118_transpose_y_1"), val = bool(true)]; + tensor var_4118 = matmul(transpose_x = var_4118_transpose_x_1, transpose_y = var_4118_transpose_y_1, x = query_states_25, y = key_states_27_cast_fp16)[name = string("op_4118")]; + fp16 var_4119_to_fp16 = const()[name = string("op_4119_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_37_cast_fp16 = mul(x = var_4118, y = var_4119_to_fp16)[name = string("attn_weights_37_cast_fp16")]; + tensor attn_weights_39_cast_fp16 = add(x = attn_weights_37_cast_fp16, y = causal_mask)[name = string("attn_weights_39_cast_fp16")]; + int32 var_4154 = const()[name = string("op_4154"), val = int32(-1)]; + tensor attn_weights_41_cast_fp16 = softmax(axis = var_4154, x = attn_weights_39_cast_fp16)[name = string("attn_weights_41_cast_fp16")]; + bool attn_output_61_transpose_x_0 = const()[name = string("attn_output_61_transpose_x_0"), val = bool(false)]; + bool attn_output_61_transpose_y_0 = const()[name = string("attn_output_61_transpose_y_0"), val = bool(false)]; + tensor attn_output_61_cast_fp16 = matmul(transpose_x = attn_output_61_transpose_x_0, transpose_y = attn_output_61_transpose_y_0, x = attn_weights_41_cast_fp16, y = value_states_39_cast_fp16)[name = string("attn_output_61_cast_fp16")]; + tensor var_4165_perm_0 = const()[name = string("op_4165_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_4169 = const()[name = string("op_4169"), val = tensor([1, 1, 2048])]; + tensor var_4165_cast_fp16 = transpose(perm = var_4165_perm_0, x = attn_output_61_cast_fp16)[name = string("transpose_46")]; + tensor attn_output_65_cast_fp16 = reshape(shape = var_4169, x = var_4165_cast_fp16)[name = string("attn_output_65_cast_fp16")]; + tensor var_4174 = const()[name = string("op_4174"), val = tensor([0, 2, 1])]; + string var_4190_pad_type_0 = const()[name = string("op_4190_pad_type_0"), val = string("valid")]; + int32 var_4190_groups_0 = const()[name = string("op_4190_groups_0"), val = int32(1)]; + tensor var_4190_strides_0 = const()[name = string("op_4190_strides_0"), val = tensor([1])]; + tensor var_4190_pad_0 = const()[name = string("op_4190_pad_0"), val = tensor([0, 0])]; + tensor var_4190_dilations_0 = const()[name = string("op_4190_dilations_0"), val = tensor([1])]; + tensor squeeze_6_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689508992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693703360))))[name = string("squeeze_6_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_4175_cast_fp16 = transpose(perm = var_4174, x = attn_output_65_cast_fp16)[name = string("transpose_45")]; + tensor var_4190_cast_fp16 = conv(dilations = var_4190_dilations_0, groups = var_4190_groups_0, pad = var_4190_pad_0, pad_type = var_4190_pad_type_0, strides = var_4190_strides_0, weight = squeeze_6_cast_fp16_to_fp32_to_fp16_palettized, x = var_4175_cast_fp16)[name = string("op_4190_cast_fp16")]; + tensor var_4194 = const()[name = string("op_4194"), val = tensor([0, 2, 1])]; + tensor attn_output_69_cast_fp16 = transpose(perm = var_4194, x = var_4190_cast_fp16)[name = string("transpose_44")]; + tensor hidden_states_69_cast_fp16 = add(x = hidden_states_61_cast_fp16, y = attn_output_69_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; + int32 var_4207 = const()[name = string("op_4207"), val = int32(-1)]; + fp16 const_206_promoted_to_fp16 = const()[name = string("const_206_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4209_cast_fp16 = mul(x = hidden_states_69_cast_fp16, y = const_206_promoted_to_fp16)[name = string("op_4209_cast_fp16")]; + bool input_119_interleave_0 = const()[name = string("input_119_interleave_0"), val = bool(false)]; + tensor input_119_cast_fp16 = concat(axis = var_4207, interleave = input_119_interleave_0, values = (hidden_states_69_cast_fp16, var_4209_cast_fp16))[name = string("input_119_cast_fp16")]; + tensor normed_109_axes_0 = const()[name = string("normed_109_axes_0"), val = tensor([-1])]; + fp16 var_4204_to_fp16 = const()[name = string("op_4204_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_109_cast_fp16 = layer_norm(axes = normed_109_axes_0, epsilon = var_4204_to_fp16, x = input_119_cast_fp16)[name = string("normed_109_cast_fp16")]; + tensor normed_111_begin_0 = const()[name = string("normed_111_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_111_end_0 = const()[name = string("normed_111_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_111_end_mask_0 = const()[name = string("normed_111_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_111_cast_fp16 = slice_by_index(begin = normed_111_begin_0, end = normed_111_end_0, end_mask = normed_111_end_mask_0, x = normed_109_cast_fp16)[name = string("normed_111_cast_fp16")]; + tensor const_209_promoted_to_fp16 = const()[name = string("const_209_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693834496)))]; + tensor x_109_cast_fp16 = mul(x = normed_111_cast_fp16, y = const_209_promoted_to_fp16)[name = string("x_109_cast_fp16")]; + tensor var_4234 = const()[name = string("op_4234"), val = tensor([0, 2, 1])]; + tensor input_121_axes_0 = const()[name = string("input_121_axes_0"), val = tensor([2])]; + tensor var_4235 = transpose(perm = var_4234, x = x_109_cast_fp16)[name = string("transpose_43")]; + tensor input_121 = expand_dims(axes = input_121_axes_0, x = var_4235)[name = string("input_121")]; + string input_123_pad_type_0 = const()[name = string("input_123_pad_type_0"), val = string("valid")]; + tensor input_123_strides_0 = const()[name = string("input_123_strides_0"), val = tensor([1, 1])]; + tensor input_123_pad_0 = const()[name = string("input_123_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_123_dilations_0 = const()[name = string("input_123_dilations_0"), val = tensor([1, 1])]; + int32 input_123_groups_0 = const()[name = string("input_123_groups_0"), val = int32(1)]; + tensor input_123 = conv(dilations = input_123_dilations_0, groups = input_123_groups_0, pad = input_123_pad_0, pad_type = input_123_pad_type_0, strides = input_123_strides_0, weight = model_model_layers_6_mlp_gate_proj_weight_palettized, x = input_121)[name = string("input_123")]; + string b_13_pad_type_0 = const()[name = string("b_13_pad_type_0"), val = string("valid")]; + tensor b_13_strides_0 = const()[name = string("b_13_strides_0"), val = tensor([1, 1])]; + tensor b_13_pad_0 = const()[name = string("b_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_13_dilations_0 = const()[name = string("b_13_dilations_0"), val = tensor([1, 1])]; + int32 b_13_groups_0 = const()[name = string("b_13_groups_0"), val = int32(1)]; + tensor b_13 = conv(dilations = b_13_dilations_0, groups = b_13_groups_0, pad = b_13_pad_0, pad_type = b_13_pad_type_0, strides = b_13_strides_0, weight = model_model_layers_6_mlp_up_proj_weight_palettized, x = input_121)[name = string("b_13")]; + tensor c_13 = silu(x = input_123)[name = string("c_13")]; + tensor input_125 = mul(x = c_13, y = b_13)[name = string("input_125")]; + string e_13_pad_type_0 = const()[name = string("e_13_pad_type_0"), val = string("valid")]; + tensor e_13_strides_0 = const()[name = string("e_13_strides_0"), val = tensor([1, 1])]; + tensor e_13_pad_0 = const()[name = string("e_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_13_dilations_0 = const()[name = string("e_13_dilations_0"), val = tensor([1, 1])]; + int32 e_13_groups_0 = const()[name = string("e_13_groups_0"), val = int32(1)]; + tensor e_13 = conv(dilations = e_13_dilations_0, groups = e_13_groups_0, pad = e_13_pad_0, pad_type = e_13_pad_type_0, strides = e_13_strides_0, weight = model_model_layers_6_mlp_down_proj_weight_palettized, x = input_125)[name = string("e_13")]; + tensor var_4257_axes_0 = const()[name = string("op_4257_axes_0"), val = tensor([2])]; + tensor var_4257 = squeeze(axes = var_4257_axes_0, x = e_13)[name = string("op_4257")]; + tensor var_4258 = const()[name = string("op_4258"), val = tensor([0, 2, 1])]; + tensor var_4259 = transpose(perm = var_4258, x = var_4257)[name = string("transpose_42")]; + tensor hidden_states_71_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = var_4259)[name = string("hidden_states_71_cast_fp16")]; + int32 var_4271 = const()[name = string("op_4271"), val = int32(-1)]; + fp16 const_210_promoted_to_fp16 = const()[name = string("const_210_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4273_cast_fp16 = mul(x = hidden_states_71_cast_fp16, y = const_210_promoted_to_fp16)[name = string("op_4273_cast_fp16")]; + bool input_127_interleave_0 = const()[name = string("input_127_interleave_0"), val = bool(false)]; + tensor input_127_cast_fp16 = concat(axis = var_4271, interleave = input_127_interleave_0, values = (hidden_states_71_cast_fp16, var_4273_cast_fp16))[name = string("input_127_cast_fp16")]; + tensor normed_113_axes_0 = const()[name = string("normed_113_axes_0"), val = tensor([-1])]; + fp16 var_4268_to_fp16 = const()[name = string("op_4268_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_113_cast_fp16 = layer_norm(axes = normed_113_axes_0, epsilon = var_4268_to_fp16, x = input_127_cast_fp16)[name = string("normed_113_cast_fp16")]; + tensor normed_115_begin_0 = const()[name = string("normed_115_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_115_end_0 = const()[name = string("normed_115_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_115_end_mask_0 = const()[name = string("normed_115_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_115_cast_fp16 = slice_by_index(begin = normed_115_begin_0, end = normed_115_end_0, end_mask = normed_115_end_mask_0, x = normed_113_cast_fp16)[name = string("normed_115_cast_fp16")]; + tensor const_213_promoted_to_fp16 = const()[name = string("const_213_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693838656)))]; + tensor hidden_states_73_cast_fp16 = mul(x = normed_115_cast_fp16, y = const_213_promoted_to_fp16)[name = string("hidden_states_73_cast_fp16")]; + tensor var_4290 = const()[name = string("op_4290"), val = tensor([0, 2, 1])]; + tensor var_4293_axes_0 = const()[name = string("op_4293_axes_0"), val = tensor([2])]; + tensor var_4291_cast_fp16 = transpose(perm = var_4290, x = hidden_states_73_cast_fp16)[name = string("transpose_41")]; + tensor var_4293_cast_fp16 = expand_dims(axes = var_4293_axes_0, x = var_4291_cast_fp16)[name = string("op_4293_cast_fp16")]; + string var_4309_pad_type_0 = const()[name = string("op_4309_pad_type_0"), val = string("valid")]; + tensor var_4309_strides_0 = const()[name = string("op_4309_strides_0"), val = tensor([1, 1])]; + tensor var_4309_pad_0 = const()[name = string("op_4309_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4309_dilations_0 = const()[name = string("op_4309_dilations_0"), val = tensor([1, 1])]; + int32 var_4309_groups_0 = const()[name = string("op_4309_groups_0"), val = int32(1)]; + tensor var_4309 = conv(dilations = var_4309_dilations_0, groups = var_4309_groups_0, pad = var_4309_pad_0, pad_type = var_4309_pad_type_0, strides = var_4309_strides_0, weight = model_model_layers_7_self_attn_q_proj_weight_palettized, x = var_4293_cast_fp16)[name = string("op_4309")]; + tensor var_4314 = const()[name = string("op_4314"), val = tensor([1, 16, 1, 128])]; + tensor var_4315 = reshape(shape = var_4314, x = var_4309)[name = string("op_4315")]; + string var_4331_pad_type_0 = const()[name = string("op_4331_pad_type_0"), val = string("valid")]; + tensor var_4331_strides_0 = const()[name = string("op_4331_strides_0"), val = tensor([1, 1])]; + tensor var_4331_pad_0 = const()[name = string("op_4331_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4331_dilations_0 = const()[name = string("op_4331_dilations_0"), val = tensor([1, 1])]; + int32 var_4331_groups_0 = const()[name = string("op_4331_groups_0"), val = int32(1)]; + tensor var_4331 = conv(dilations = var_4331_dilations_0, groups = var_4331_groups_0, pad = var_4331_pad_0, pad_type = var_4331_pad_type_0, strides = var_4331_strides_0, weight = model_model_layers_7_self_attn_k_proj_weight_palettized, x = var_4293_cast_fp16)[name = string("op_4331")]; + tensor var_4336 = const()[name = string("op_4336"), val = tensor([1, 8, 1, 128])]; + tensor var_4337 = reshape(shape = var_4336, x = var_4331)[name = string("op_4337")]; + string var_4353_pad_type_0 = const()[name = string("op_4353_pad_type_0"), val = string("valid")]; + tensor var_4353_strides_0 = const()[name = string("op_4353_strides_0"), val = tensor([1, 1])]; + tensor var_4353_pad_0 = const()[name = string("op_4353_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4353_dilations_0 = const()[name = string("op_4353_dilations_0"), val = tensor([1, 1])]; + int32 var_4353_groups_0 = const()[name = string("op_4353_groups_0"), val = int32(1)]; + tensor var_4353 = conv(dilations = var_4353_dilations_0, groups = var_4353_groups_0, pad = var_4353_pad_0, pad_type = var_4353_pad_type_0, strides = var_4353_strides_0, weight = model_model_layers_7_self_attn_v_proj_weight_palettized, x = var_4293_cast_fp16)[name = string("op_4353")]; + tensor var_4358 = const()[name = string("op_4358"), val = tensor([1, 8, 1, 128])]; + tensor var_4359 = reshape(shape = var_4358, x = var_4353)[name = string("op_4359")]; + int32 var_4374 = const()[name = string("op_4374"), val = int32(-1)]; + fp16 const_214_promoted = const()[name = string("const_214_promoted"), val = fp16(-0x1p+0)]; + tensor var_4376 = mul(x = var_4315, y = const_214_promoted)[name = string("op_4376")]; + bool input_131_interleave_0 = const()[name = string("input_131_interleave_0"), val = bool(false)]; + tensor input_131 = concat(axis = var_4374, interleave = input_131_interleave_0, values = (var_4315, var_4376))[name = string("input_131")]; + tensor normed_117_axes_0 = const()[name = string("normed_117_axes_0"), val = tensor([-1])]; + fp16 var_4371_to_fp16 = const()[name = string("op_4371_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_117_cast_fp16 = layer_norm(axes = normed_117_axes_0, epsilon = var_4371_to_fp16, x = input_131)[name = string("normed_117_cast_fp16")]; + tensor normed_119_begin_0 = const()[name = string("normed_119_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_119_end_0 = const()[name = string("normed_119_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_119_end_mask_0 = const()[name = string("normed_119_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_119 = slice_by_index(begin = normed_119_begin_0, end = normed_119_end_0, end_mask = normed_119_end_mask_0, x = normed_117_cast_fp16)[name = string("normed_119")]; + tensor const_217 = const()[name = string("const_217"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693842816)))]; + tensor q_15 = mul(x = normed_119, y = const_217)[name = string("q_15")]; + int32 var_4399 = const()[name = string("op_4399"), val = int32(-1)]; + fp16 const_218_promoted = const()[name = string("const_218_promoted"), val = fp16(-0x1p+0)]; + tensor var_4401 = mul(x = var_4337, y = const_218_promoted)[name = string("op_4401")]; + bool input_133_interleave_0 = const()[name = string("input_133_interleave_0"), val = bool(false)]; + tensor input_133 = concat(axis = var_4399, interleave = input_133_interleave_0, values = (var_4337, var_4401))[name = string("input_133")]; + tensor normed_121_axes_0 = const()[name = string("normed_121_axes_0"), val = tensor([-1])]; + fp16 var_4396_to_fp16 = const()[name = string("op_4396_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_121_cast_fp16 = layer_norm(axes = normed_121_axes_0, epsilon = var_4396_to_fp16, x = input_133)[name = string("normed_121_cast_fp16")]; + tensor normed_123_begin_0 = const()[name = string("normed_123_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_123_end_0 = const()[name = string("normed_123_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_123_end_mask_0 = const()[name = string("normed_123_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_123 = slice_by_index(begin = normed_123_begin_0, end = normed_123_end_0, end_mask = normed_123_end_mask_0, x = normed_121_cast_fp16)[name = string("normed_123")]; + tensor const_221 = const()[name = string("const_221"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693843136)))]; + tensor k_15 = mul(x = normed_123, y = const_221)[name = string("k_15")]; + tensor var_4415 = mul(x = q_15, y = cos_1_cast_fp16)[name = string("op_4415")]; + tensor x1_29_begin_0 = const()[name = string("x1_29_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_29_end_0 = const()[name = string("x1_29_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_29_end_mask_0 = const()[name = string("x1_29_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_29 = slice_by_index(begin = x1_29_begin_0, end = x1_29_end_0, end_mask = x1_29_end_mask_0, x = q_15)[name = string("x1_29")]; + tensor x2_29_begin_0 = const()[name = string("x2_29_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_29_end_0 = const()[name = string("x2_29_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_29_end_mask_0 = const()[name = string("x2_29_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_29 = slice_by_index(begin = x2_29_begin_0, end = x2_29_end_0, end_mask = x2_29_end_mask_0, x = q_15)[name = string("x2_29")]; + fp16 const_224_promoted = const()[name = string("const_224_promoted"), val = fp16(-0x1p+0)]; + tensor var_4436 = mul(x = x2_29, y = const_224_promoted)[name = string("op_4436")]; + int32 var_4438 = const()[name = string("op_4438"), val = int32(-1)]; + bool var_4439_interleave_0 = const()[name = string("op_4439_interleave_0"), val = bool(false)]; + tensor var_4439 = concat(axis = var_4438, interleave = var_4439_interleave_0, values = (var_4436, x1_29))[name = string("op_4439")]; + tensor var_4440 = mul(x = var_4439, y = sin_1_cast_fp16)[name = string("op_4440")]; + tensor query_states_29 = add(x = var_4415, y = var_4440)[name = string("query_states_29")]; + tensor var_4443 = mul(x = k_15, y = cos_1_cast_fp16)[name = string("op_4443")]; + tensor x1_31_begin_0 = const()[name = string("x1_31_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_31_end_0 = const()[name = string("x1_31_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_31_end_mask_0 = const()[name = string("x1_31_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_31 = slice_by_index(begin = x1_31_begin_0, end = x1_31_end_0, end_mask = x1_31_end_mask_0, x = k_15)[name = string("x1_31")]; + tensor x2_31_begin_0 = const()[name = string("x2_31_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_31_end_0 = const()[name = string("x2_31_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_31_end_mask_0 = const()[name = string("x2_31_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_31 = slice_by_index(begin = x2_31_begin_0, end = x2_31_end_0, end_mask = x2_31_end_mask_0, x = k_15)[name = string("x2_31")]; + fp16 const_227_promoted = const()[name = string("const_227_promoted"), val = fp16(-0x1p+0)]; + tensor var_4464 = mul(x = x2_31, y = const_227_promoted)[name = string("op_4464")]; + int32 var_4466 = const()[name = string("op_4466"), val = int32(-1)]; + bool var_4467_interleave_0 = const()[name = string("op_4467_interleave_0"), val = bool(false)]; + tensor var_4467 = concat(axis = var_4466, interleave = var_4467_interleave_0, values = (var_4464, x1_31))[name = string("op_4467")]; + tensor var_4468 = mul(x = var_4467, y = sin_1_cast_fp16)[name = string("op_4468")]; + tensor key_states_29 = add(x = var_4443, y = var_4468)[name = string("key_states_29")]; + tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([7])]; + tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; + tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; + tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([8])]; + int32 concat_58_axis_0 = const()[name = string("concat_58_axis_0"), val = int32(0)]; + bool concat_58_interleave_0 = const()[name = string("concat_58_interleave_0"), val = bool(false)]; + tensor concat_58 = concat(axis = concat_58_axis_0, interleave = concat_58_interleave_0, values = (expand_dims_84, expand_dims_85, current_pos, expand_dims_87))[name = string("concat_58")]; + tensor concat_59_values1_0 = const()[name = string("concat_59_values1_0"), val = tensor([0])]; + tensor concat_59_values3_0 = const()[name = string("concat_59_values3_0"), val = tensor([0])]; + int32 concat_59_axis_0 = const()[name = string("concat_59_axis_0"), val = int32(0)]; + bool concat_59_interleave_0 = const()[name = string("concat_59_interleave_0"), val = bool(false)]; + tensor concat_59 = concat(axis = concat_59_axis_0, interleave = concat_59_interleave_0, values = (expand_dims_88, concat_59_values1_0, var_1001, concat_59_values3_0))[name = string("concat_59")]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_58, begin_mask = model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0, end = concat_59, end_mask = model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_15_stride_0, update = key_states_29, x = coreml_update_state_41)[name = string("model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_70_write_state")]; + tensor coreml_update_state_42 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_70")]; + tensor expand_dims_90 = const()[name = string("expand_dims_90"), val = tensor([35])]; + tensor expand_dims_91 = const()[name = string("expand_dims_91"), val = tensor([0])]; + tensor expand_dims_93 = const()[name = string("expand_dims_93"), val = tensor([0])]; + tensor expand_dims_94 = const()[name = string("expand_dims_94"), val = tensor([36])]; + int32 concat_62_axis_0 = const()[name = string("concat_62_axis_0"), val = int32(0)]; + bool concat_62_interleave_0 = const()[name = string("concat_62_interleave_0"), val = bool(false)]; + tensor concat_62 = concat(axis = concat_62_axis_0, interleave = concat_62_interleave_0, values = (expand_dims_90, expand_dims_91, current_pos, expand_dims_93))[name = string("concat_62")]; + tensor concat_63_values1_0 = const()[name = string("concat_63_values1_0"), val = tensor([0])]; + tensor concat_63_values3_0 = const()[name = string("concat_63_values3_0"), val = tensor([0])]; + int32 concat_63_axis_0 = const()[name = string("concat_63_axis_0"), val = int32(0)]; + bool concat_63_interleave_0 = const()[name = string("concat_63_interleave_0"), val = bool(false)]; + tensor concat_63 = concat(axis = concat_63_axis_0, interleave = concat_63_interleave_0, values = (expand_dims_94, concat_63_values1_0, var_1001, concat_63_values3_0))[name = string("concat_63")]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_62, begin_mask = model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0, end = concat_63, end_mask = model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_16_stride_0, update = var_4359, x = coreml_update_state_42)[name = string("model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_71_write_state")]; + tensor coreml_update_state_43 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_71")]; + tensor var_4523_begin_0 = const()[name = string("op_4523_begin_0"), val = tensor([7, 0, 0, 0])]; + tensor var_4523_end_0 = const()[name = string("op_4523_end_0"), val = tensor([8, 8, 1024, 128])]; + tensor var_4523_end_mask_0 = const()[name = string("op_4523_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4523_cast_fp16 = slice_by_index(begin = var_4523_begin_0, end = var_4523_end_0, end_mask = var_4523_end_mask_0, x = coreml_update_state_43)[name = string("op_4523_cast_fp16")]; + tensor K_layer_cache_15_axes_0 = const()[name = string("K_layer_cache_15_axes_0"), val = tensor([0])]; + tensor K_layer_cache_15_cast_fp16 = squeeze(axes = K_layer_cache_15_axes_0, x = var_4523_cast_fp16)[name = string("K_layer_cache_15_cast_fp16")]; + tensor var_4530_begin_0 = const()[name = string("op_4530_begin_0"), val = tensor([35, 0, 0, 0])]; + tensor var_4530_end_0 = const()[name = string("op_4530_end_0"), val = tensor([36, 8, 1024, 128])]; + tensor var_4530_end_mask_0 = const()[name = string("op_4530_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4530_cast_fp16 = slice_by_index(begin = var_4530_begin_0, end = var_4530_end_0, end_mask = var_4530_end_mask_0, x = coreml_update_state_43)[name = string("op_4530_cast_fp16")]; + tensor V_layer_cache_15_axes_0 = const()[name = string("V_layer_cache_15_axes_0"), val = tensor([0])]; + tensor V_layer_cache_15_cast_fp16 = squeeze(axes = V_layer_cache_15_axes_0, x = var_4530_cast_fp16)[name = string("V_layer_cache_15_cast_fp16")]; + tensor x_115_axes_0 = const()[name = string("x_115_axes_0"), val = tensor([1])]; + tensor x_115_cast_fp16 = expand_dims(axes = x_115_axes_0, x = K_layer_cache_15_cast_fp16)[name = string("x_115_cast_fp16")]; + tensor var_4567 = const()[name = string("op_4567"), val = tensor([1, 2, 1, 1])]; + tensor x_117_cast_fp16 = tile(reps = var_4567, x = x_115_cast_fp16)[name = string("x_117_cast_fp16")]; + tensor var_4579 = const()[name = string("op_4579"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_31_cast_fp16 = reshape(shape = var_4579, x = x_117_cast_fp16)[name = string("key_states_31_cast_fp16")]; + tensor x_121_axes_0 = const()[name = string("x_121_axes_0"), val = tensor([1])]; + tensor x_121_cast_fp16 = expand_dims(axes = x_121_axes_0, x = V_layer_cache_15_cast_fp16)[name = string("x_121_cast_fp16")]; + tensor var_4587 = const()[name = string("op_4587"), val = tensor([1, 2, 1, 1])]; + tensor x_123_cast_fp16 = tile(reps = var_4587, x = x_121_cast_fp16)[name = string("x_123_cast_fp16")]; + tensor var_4599 = const()[name = string("op_4599"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_45_cast_fp16 = reshape(shape = var_4599, x = x_123_cast_fp16)[name = string("value_states_45_cast_fp16")]; + bool var_4614_transpose_x_1 = const()[name = string("op_4614_transpose_x_1"), val = bool(false)]; + bool var_4614_transpose_y_1 = const()[name = string("op_4614_transpose_y_1"), val = bool(true)]; + tensor var_4614 = matmul(transpose_x = var_4614_transpose_x_1, transpose_y = var_4614_transpose_y_1, x = query_states_29, y = key_states_31_cast_fp16)[name = string("op_4614")]; + fp16 var_4615_to_fp16 = const()[name = string("op_4615_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_43_cast_fp16 = mul(x = var_4614, y = var_4615_to_fp16)[name = string("attn_weights_43_cast_fp16")]; + tensor attn_weights_45_cast_fp16 = add(x = attn_weights_43_cast_fp16, y = causal_mask)[name = string("attn_weights_45_cast_fp16")]; + int32 var_4650 = const()[name = string("op_4650"), val = int32(-1)]; + tensor attn_weights_47_cast_fp16 = softmax(axis = var_4650, x = attn_weights_45_cast_fp16)[name = string("attn_weights_47_cast_fp16")]; + bool attn_output_71_transpose_x_0 = const()[name = string("attn_output_71_transpose_x_0"), val = bool(false)]; + bool attn_output_71_transpose_y_0 = const()[name = string("attn_output_71_transpose_y_0"), val = bool(false)]; + tensor attn_output_71_cast_fp16 = matmul(transpose_x = attn_output_71_transpose_x_0, transpose_y = attn_output_71_transpose_y_0, x = attn_weights_47_cast_fp16, y = value_states_45_cast_fp16)[name = string("attn_output_71_cast_fp16")]; + tensor var_4661_perm_0 = const()[name = string("op_4661_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_4665 = const()[name = string("op_4665"), val = tensor([1, 1, 2048])]; + tensor var_4661_cast_fp16 = transpose(perm = var_4661_perm_0, x = attn_output_71_cast_fp16)[name = string("transpose_40")]; + tensor attn_output_75_cast_fp16 = reshape(shape = var_4665, x = var_4661_cast_fp16)[name = string("attn_output_75_cast_fp16")]; + tensor var_4670 = const()[name = string("op_4670"), val = tensor([0, 2, 1])]; + string var_4686_pad_type_0 = const()[name = string("op_4686_pad_type_0"), val = string("valid")]; + int32 var_4686_groups_0 = const()[name = string("op_4686_groups_0"), val = int32(1)]; + tensor var_4686_strides_0 = const()[name = string("op_4686_strides_0"), val = tensor([1])]; + tensor var_4686_pad_0 = const()[name = string("op_4686_pad_0"), val = tensor([0, 0])]; + tensor var_4686_dilations_0 = const()[name = string("op_4686_dilations_0"), val = tensor([1])]; + tensor squeeze_7_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693843456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698037824))))[name = string("squeeze_7_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_4671_cast_fp16 = transpose(perm = var_4670, x = attn_output_75_cast_fp16)[name = string("transpose_39")]; + tensor var_4686_cast_fp16 = conv(dilations = var_4686_dilations_0, groups = var_4686_groups_0, pad = var_4686_pad_0, pad_type = var_4686_pad_type_0, strides = var_4686_strides_0, weight = squeeze_7_cast_fp16_to_fp32_to_fp16_palettized, x = var_4671_cast_fp16)[name = string("op_4686_cast_fp16")]; + tensor var_4690 = const()[name = string("op_4690"), val = tensor([0, 2, 1])]; + tensor attn_output_79_cast_fp16 = transpose(perm = var_4690, x = var_4686_cast_fp16)[name = string("transpose_38")]; + tensor hidden_states_79_cast_fp16 = add(x = hidden_states_71_cast_fp16, y = attn_output_79_cast_fp16)[name = string("hidden_states_79_cast_fp16")]; + int32 var_4703 = const()[name = string("op_4703"), val = int32(-1)]; + fp16 const_236_promoted_to_fp16 = const()[name = string("const_236_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4705_cast_fp16 = mul(x = hidden_states_79_cast_fp16, y = const_236_promoted_to_fp16)[name = string("op_4705_cast_fp16")]; + bool input_137_interleave_0 = const()[name = string("input_137_interleave_0"), val = bool(false)]; + tensor input_137_cast_fp16 = concat(axis = var_4703, interleave = input_137_interleave_0, values = (hidden_states_79_cast_fp16, var_4705_cast_fp16))[name = string("input_137_cast_fp16")]; + tensor normed_125_axes_0 = const()[name = string("normed_125_axes_0"), val = tensor([-1])]; + fp16 var_4700_to_fp16 = const()[name = string("op_4700_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_125_cast_fp16 = layer_norm(axes = normed_125_axes_0, epsilon = var_4700_to_fp16, x = input_137_cast_fp16)[name = string("normed_125_cast_fp16")]; + tensor normed_127_begin_0 = const()[name = string("normed_127_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_127_end_0 = const()[name = string("normed_127_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_127_end_mask_0 = const()[name = string("normed_127_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_127_cast_fp16 = slice_by_index(begin = normed_127_begin_0, end = normed_127_end_0, end_mask = normed_127_end_mask_0, x = normed_125_cast_fp16)[name = string("normed_127_cast_fp16")]; + tensor const_239_promoted_to_fp16 = const()[name = string("const_239_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698168960)))]; + tensor x_125_cast_fp16 = mul(x = normed_127_cast_fp16, y = const_239_promoted_to_fp16)[name = string("x_125_cast_fp16")]; + tensor var_4730 = const()[name = string("op_4730"), val = tensor([0, 2, 1])]; + tensor input_139_axes_0 = const()[name = string("input_139_axes_0"), val = tensor([2])]; + tensor var_4731 = transpose(perm = var_4730, x = x_125_cast_fp16)[name = string("transpose_37")]; + tensor input_139 = expand_dims(axes = input_139_axes_0, x = var_4731)[name = string("input_139")]; + string input_141_pad_type_0 = const()[name = string("input_141_pad_type_0"), val = string("valid")]; + tensor input_141_strides_0 = const()[name = string("input_141_strides_0"), val = tensor([1, 1])]; + tensor input_141_pad_0 = const()[name = string("input_141_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_141_dilations_0 = const()[name = string("input_141_dilations_0"), val = tensor([1, 1])]; + int32 input_141_groups_0 = const()[name = string("input_141_groups_0"), val = int32(1)]; + tensor input_141 = conv(dilations = input_141_dilations_0, groups = input_141_groups_0, pad = input_141_pad_0, pad_type = input_141_pad_type_0, strides = input_141_strides_0, weight = model_model_layers_7_mlp_gate_proj_weight_palettized, x = input_139)[name = string("input_141")]; + string b_15_pad_type_0 = const()[name = string("b_15_pad_type_0"), val = string("valid")]; + tensor b_15_strides_0 = const()[name = string("b_15_strides_0"), val = tensor([1, 1])]; + tensor b_15_pad_0 = const()[name = string("b_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_15_dilations_0 = const()[name = string("b_15_dilations_0"), val = tensor([1, 1])]; + int32 b_15_groups_0 = const()[name = string("b_15_groups_0"), val = int32(1)]; + tensor b_15 = conv(dilations = b_15_dilations_0, groups = b_15_groups_0, pad = b_15_pad_0, pad_type = b_15_pad_type_0, strides = b_15_strides_0, weight = model_model_layers_7_mlp_up_proj_weight_palettized, x = input_139)[name = string("b_15")]; + tensor c_15 = silu(x = input_141)[name = string("c_15")]; + tensor input_143 = mul(x = c_15, y = b_15)[name = string("input_143")]; + string e_15_pad_type_0 = const()[name = string("e_15_pad_type_0"), val = string("valid")]; + tensor e_15_strides_0 = const()[name = string("e_15_strides_0"), val = tensor([1, 1])]; + tensor e_15_pad_0 = const()[name = string("e_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_15_dilations_0 = const()[name = string("e_15_dilations_0"), val = tensor([1, 1])]; + int32 e_15_groups_0 = const()[name = string("e_15_groups_0"), val = int32(1)]; + tensor e_15 = conv(dilations = e_15_dilations_0, groups = e_15_groups_0, pad = e_15_pad_0, pad_type = e_15_pad_type_0, strides = e_15_strides_0, weight = model_model_layers_7_mlp_down_proj_weight_palettized, x = input_143)[name = string("e_15")]; + tensor var_4753_axes_0 = const()[name = string("op_4753_axes_0"), val = tensor([2])]; + tensor var_4753 = squeeze(axes = var_4753_axes_0, x = e_15)[name = string("op_4753")]; + tensor var_4754 = const()[name = string("op_4754"), val = tensor([0, 2, 1])]; + tensor var_4755 = transpose(perm = var_4754, x = var_4753)[name = string("transpose_36")]; + tensor hidden_states_81_cast_fp16 = add(x = hidden_states_79_cast_fp16, y = var_4755)[name = string("hidden_states_81_cast_fp16")]; + int32 var_4767 = const()[name = string("op_4767"), val = int32(-1)]; + fp16 const_240_promoted_to_fp16 = const()[name = string("const_240_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4769_cast_fp16 = mul(x = hidden_states_81_cast_fp16, y = const_240_promoted_to_fp16)[name = string("op_4769_cast_fp16")]; + bool input_145_interleave_0 = const()[name = string("input_145_interleave_0"), val = bool(false)]; + tensor input_145_cast_fp16 = concat(axis = var_4767, interleave = input_145_interleave_0, values = (hidden_states_81_cast_fp16, var_4769_cast_fp16))[name = string("input_145_cast_fp16")]; + tensor normed_129_axes_0 = const()[name = string("normed_129_axes_0"), val = tensor([-1])]; + fp16 var_4764_to_fp16 = const()[name = string("op_4764_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_129_cast_fp16 = layer_norm(axes = normed_129_axes_0, epsilon = var_4764_to_fp16, x = input_145_cast_fp16)[name = string("normed_129_cast_fp16")]; + tensor normed_131_begin_0 = const()[name = string("normed_131_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_131_end_0 = const()[name = string("normed_131_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_131_end_mask_0 = const()[name = string("normed_131_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_131_cast_fp16 = slice_by_index(begin = normed_131_begin_0, end = normed_131_end_0, end_mask = normed_131_end_mask_0, x = normed_129_cast_fp16)[name = string("normed_131_cast_fp16")]; + tensor const_243_promoted_to_fp16 = const()[name = string("const_243_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698173120)))]; + tensor hidden_states_83_cast_fp16 = mul(x = normed_131_cast_fp16, y = const_243_promoted_to_fp16)[name = string("hidden_states_83_cast_fp16")]; + tensor var_4786 = const()[name = string("op_4786"), val = tensor([0, 2, 1])]; + tensor var_4789_axes_0 = const()[name = string("op_4789_axes_0"), val = tensor([2])]; + tensor var_4787_cast_fp16 = transpose(perm = var_4786, x = hidden_states_83_cast_fp16)[name = string("transpose_35")]; + tensor var_4789_cast_fp16 = expand_dims(axes = var_4789_axes_0, x = var_4787_cast_fp16)[name = string("op_4789_cast_fp16")]; + string var_4805_pad_type_0 = const()[name = string("op_4805_pad_type_0"), val = string("valid")]; + tensor var_4805_strides_0 = const()[name = string("op_4805_strides_0"), val = tensor([1, 1])]; + tensor var_4805_pad_0 = const()[name = string("op_4805_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4805_dilations_0 = const()[name = string("op_4805_dilations_0"), val = tensor([1, 1])]; + int32 var_4805_groups_0 = const()[name = string("op_4805_groups_0"), val = int32(1)]; + tensor var_4805 = conv(dilations = var_4805_dilations_0, groups = var_4805_groups_0, pad = var_4805_pad_0, pad_type = var_4805_pad_type_0, strides = var_4805_strides_0, weight = model_model_layers_8_self_attn_q_proj_weight_palettized, x = var_4789_cast_fp16)[name = string("op_4805")]; + tensor var_4810 = const()[name = string("op_4810"), val = tensor([1, 16, 1, 128])]; + tensor var_4811 = reshape(shape = var_4810, x = var_4805)[name = string("op_4811")]; + string var_4827_pad_type_0 = const()[name = string("op_4827_pad_type_0"), val = string("valid")]; + tensor var_4827_strides_0 = const()[name = string("op_4827_strides_0"), val = tensor([1, 1])]; + tensor var_4827_pad_0 = const()[name = string("op_4827_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4827_dilations_0 = const()[name = string("op_4827_dilations_0"), val = tensor([1, 1])]; + int32 var_4827_groups_0 = const()[name = string("op_4827_groups_0"), val = int32(1)]; + tensor var_4827 = conv(dilations = var_4827_dilations_0, groups = var_4827_groups_0, pad = var_4827_pad_0, pad_type = var_4827_pad_type_0, strides = var_4827_strides_0, weight = model_model_layers_8_self_attn_k_proj_weight_palettized, x = var_4789_cast_fp16)[name = string("op_4827")]; + tensor var_4832 = const()[name = string("op_4832"), val = tensor([1, 8, 1, 128])]; + tensor var_4833 = reshape(shape = var_4832, x = var_4827)[name = string("op_4833")]; + string var_4849_pad_type_0 = const()[name = string("op_4849_pad_type_0"), val = string("valid")]; + tensor var_4849_strides_0 = const()[name = string("op_4849_strides_0"), val = tensor([1, 1])]; + tensor var_4849_pad_0 = const()[name = string("op_4849_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4849_dilations_0 = const()[name = string("op_4849_dilations_0"), val = tensor([1, 1])]; + int32 var_4849_groups_0 = const()[name = string("op_4849_groups_0"), val = int32(1)]; + tensor var_4849 = conv(dilations = var_4849_dilations_0, groups = var_4849_groups_0, pad = var_4849_pad_0, pad_type = var_4849_pad_type_0, strides = var_4849_strides_0, weight = model_model_layers_8_self_attn_v_proj_weight_palettized, x = var_4789_cast_fp16)[name = string("op_4849")]; + tensor var_4854 = const()[name = string("op_4854"), val = tensor([1, 8, 1, 128])]; + tensor var_4855 = reshape(shape = var_4854, x = var_4849)[name = string("op_4855")]; + int32 var_4870 = const()[name = string("op_4870"), val = int32(-1)]; + fp16 const_244_promoted = const()[name = string("const_244_promoted"), val = fp16(-0x1p+0)]; + tensor var_4872 = mul(x = var_4811, y = const_244_promoted)[name = string("op_4872")]; + bool input_149_interleave_0 = const()[name = string("input_149_interleave_0"), val = bool(false)]; + tensor input_149 = concat(axis = var_4870, interleave = input_149_interleave_0, values = (var_4811, var_4872))[name = string("input_149")]; + tensor normed_133_axes_0 = const()[name = string("normed_133_axes_0"), val = tensor([-1])]; + fp16 var_4867_to_fp16 = const()[name = string("op_4867_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_133_cast_fp16 = layer_norm(axes = normed_133_axes_0, epsilon = var_4867_to_fp16, x = input_149)[name = string("normed_133_cast_fp16")]; + tensor normed_135_begin_0 = const()[name = string("normed_135_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_135_end_0 = const()[name = string("normed_135_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_135_end_mask_0 = const()[name = string("normed_135_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_135 = slice_by_index(begin = normed_135_begin_0, end = normed_135_end_0, end_mask = normed_135_end_mask_0, x = normed_133_cast_fp16)[name = string("normed_135")]; + tensor const_247 = const()[name = string("const_247"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698177280)))]; + tensor q_17 = mul(x = normed_135, y = const_247)[name = string("q_17")]; + int32 var_4895 = const()[name = string("op_4895"), val = int32(-1)]; + fp16 const_248_promoted = const()[name = string("const_248_promoted"), val = fp16(-0x1p+0)]; + tensor var_4897 = mul(x = var_4833, y = const_248_promoted)[name = string("op_4897")]; + bool input_151_interleave_0 = const()[name = string("input_151_interleave_0"), val = bool(false)]; + tensor input_151 = concat(axis = var_4895, interleave = input_151_interleave_0, values = (var_4833, var_4897))[name = string("input_151")]; + tensor normed_137_axes_0 = const()[name = string("normed_137_axes_0"), val = tensor([-1])]; + fp16 var_4892_to_fp16 = const()[name = string("op_4892_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_137_cast_fp16 = layer_norm(axes = normed_137_axes_0, epsilon = var_4892_to_fp16, x = input_151)[name = string("normed_137_cast_fp16")]; + tensor normed_139_begin_0 = const()[name = string("normed_139_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_139_end_0 = const()[name = string("normed_139_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_139_end_mask_0 = const()[name = string("normed_139_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_139 = slice_by_index(begin = normed_139_begin_0, end = normed_139_end_0, end_mask = normed_139_end_mask_0, x = normed_137_cast_fp16)[name = string("normed_139")]; + tensor const_251 = const()[name = string("const_251"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698177600)))]; + tensor k_17 = mul(x = normed_139, y = const_251)[name = string("k_17")]; + tensor var_4911 = mul(x = q_17, y = cos_1_cast_fp16)[name = string("op_4911")]; + tensor x1_33_begin_0 = const()[name = string("x1_33_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_33_end_0 = const()[name = string("x1_33_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_33_end_mask_0 = const()[name = string("x1_33_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_33 = slice_by_index(begin = x1_33_begin_0, end = x1_33_end_0, end_mask = x1_33_end_mask_0, x = q_17)[name = string("x1_33")]; + tensor x2_33_begin_0 = const()[name = string("x2_33_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_33_end_0 = const()[name = string("x2_33_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_33_end_mask_0 = const()[name = string("x2_33_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_33 = slice_by_index(begin = x2_33_begin_0, end = x2_33_end_0, end_mask = x2_33_end_mask_0, x = q_17)[name = string("x2_33")]; + fp16 const_254_promoted = const()[name = string("const_254_promoted"), val = fp16(-0x1p+0)]; + tensor var_4932 = mul(x = x2_33, y = const_254_promoted)[name = string("op_4932")]; + int32 var_4934 = const()[name = string("op_4934"), val = int32(-1)]; + bool var_4935_interleave_0 = const()[name = string("op_4935_interleave_0"), val = bool(false)]; + tensor var_4935 = concat(axis = var_4934, interleave = var_4935_interleave_0, values = (var_4932, x1_33))[name = string("op_4935")]; + tensor var_4936 = mul(x = var_4935, y = sin_1_cast_fp16)[name = string("op_4936")]; + tensor query_states_33 = add(x = var_4911, y = var_4936)[name = string("query_states_33")]; + tensor var_4939 = mul(x = k_17, y = cos_1_cast_fp16)[name = string("op_4939")]; + tensor x1_35_begin_0 = const()[name = string("x1_35_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_35_end_0 = const()[name = string("x1_35_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_35_end_mask_0 = const()[name = string("x1_35_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_35 = slice_by_index(begin = x1_35_begin_0, end = x1_35_end_0, end_mask = x1_35_end_mask_0, x = k_17)[name = string("x1_35")]; + tensor x2_35_begin_0 = const()[name = string("x2_35_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_35_end_0 = const()[name = string("x2_35_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_35_end_mask_0 = const()[name = string("x2_35_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_35 = slice_by_index(begin = x2_35_begin_0, end = x2_35_end_0, end_mask = x2_35_end_mask_0, x = k_17)[name = string("x2_35")]; + fp16 const_257_promoted = const()[name = string("const_257_promoted"), val = fp16(-0x1p+0)]; + tensor var_4960 = mul(x = x2_35, y = const_257_promoted)[name = string("op_4960")]; + int32 var_4962 = const()[name = string("op_4962"), val = int32(-1)]; + bool var_4963_interleave_0 = const()[name = string("op_4963_interleave_0"), val = bool(false)]; + tensor var_4963 = concat(axis = var_4962, interleave = var_4963_interleave_0, values = (var_4960, x1_35))[name = string("op_4963")]; + tensor var_4964 = mul(x = var_4963, y = sin_1_cast_fp16)[name = string("op_4964")]; + tensor key_states_33 = add(x = var_4939, y = var_4964)[name = string("key_states_33")]; + tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([8])]; + tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; + tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; + tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([9])]; + int32 concat_66_axis_0 = const()[name = string("concat_66_axis_0"), val = int32(0)]; + bool concat_66_interleave_0 = const()[name = string("concat_66_interleave_0"), val = bool(false)]; + tensor concat_66 = concat(axis = concat_66_axis_0, interleave = concat_66_interleave_0, values = (expand_dims_96, expand_dims_97, current_pos, expand_dims_99))[name = string("concat_66")]; + tensor concat_67_values1_0 = const()[name = string("concat_67_values1_0"), val = tensor([0])]; + tensor concat_67_values3_0 = const()[name = string("concat_67_values3_0"), val = tensor([0])]; + int32 concat_67_axis_0 = const()[name = string("concat_67_axis_0"), val = int32(0)]; + bool concat_67_interleave_0 = const()[name = string("concat_67_interleave_0"), val = bool(false)]; + tensor concat_67 = concat(axis = concat_67_axis_0, interleave = concat_67_interleave_0, values = (expand_dims_100, concat_67_values1_0, var_1001, concat_67_values3_0))[name = string("concat_67")]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_66, begin_mask = model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0, end = concat_67, end_mask = model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_17_stride_0, update = key_states_33, x = coreml_update_state_43)[name = string("model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_72_write_state")]; + tensor coreml_update_state_44 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_72")]; + tensor expand_dims_102 = const()[name = string("expand_dims_102"), val = tensor([36])]; + tensor expand_dims_103 = const()[name = string("expand_dims_103"), val = tensor([0])]; + tensor expand_dims_105 = const()[name = string("expand_dims_105"), val = tensor([0])]; + tensor expand_dims_106 = const()[name = string("expand_dims_106"), val = tensor([37])]; + int32 concat_70_axis_0 = const()[name = string("concat_70_axis_0"), val = int32(0)]; + bool concat_70_interleave_0 = const()[name = string("concat_70_interleave_0"), val = bool(false)]; + tensor concat_70 = concat(axis = concat_70_axis_0, interleave = concat_70_interleave_0, values = (expand_dims_102, expand_dims_103, current_pos, expand_dims_105))[name = string("concat_70")]; + tensor concat_71_values1_0 = const()[name = string("concat_71_values1_0"), val = tensor([0])]; + tensor concat_71_values3_0 = const()[name = string("concat_71_values3_0"), val = tensor([0])]; + int32 concat_71_axis_0 = const()[name = string("concat_71_axis_0"), val = int32(0)]; + bool concat_71_interleave_0 = const()[name = string("concat_71_interleave_0"), val = bool(false)]; + tensor concat_71 = concat(axis = concat_71_axis_0, interleave = concat_71_interleave_0, values = (expand_dims_106, concat_71_values1_0, var_1001, concat_71_values3_0))[name = string("concat_71")]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_70, begin_mask = model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0, end = concat_71, end_mask = model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_18_stride_0, update = var_4855, x = coreml_update_state_44)[name = string("model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_73_write_state")]; + tensor coreml_update_state_45 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_73")]; + tensor var_5019_begin_0 = const()[name = string("op_5019_begin_0"), val = tensor([8, 0, 0, 0])]; + tensor var_5019_end_0 = const()[name = string("op_5019_end_0"), val = tensor([9, 8, 1024, 128])]; + tensor var_5019_end_mask_0 = const()[name = string("op_5019_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_5019_cast_fp16 = slice_by_index(begin = var_5019_begin_0, end = var_5019_end_0, end_mask = var_5019_end_mask_0, x = coreml_update_state_45)[name = string("op_5019_cast_fp16")]; + tensor K_layer_cache_17_axes_0 = const()[name = string("K_layer_cache_17_axes_0"), val = tensor([0])]; + tensor K_layer_cache_17_cast_fp16 = squeeze(axes = K_layer_cache_17_axes_0, x = var_5019_cast_fp16)[name = string("K_layer_cache_17_cast_fp16")]; + tensor var_5026_begin_0 = const()[name = string("op_5026_begin_0"), val = tensor([36, 0, 0, 0])]; + tensor var_5026_end_0 = const()[name = string("op_5026_end_0"), val = tensor([37, 8, 1024, 128])]; + tensor var_5026_end_mask_0 = const()[name = string("op_5026_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_5026_cast_fp16 = slice_by_index(begin = var_5026_begin_0, end = var_5026_end_0, end_mask = var_5026_end_mask_0, x = coreml_update_state_45)[name = string("op_5026_cast_fp16")]; + tensor V_layer_cache_17_axes_0 = const()[name = string("V_layer_cache_17_axes_0"), val = tensor([0])]; + tensor V_layer_cache_17_cast_fp16 = squeeze(axes = V_layer_cache_17_axes_0, x = var_5026_cast_fp16)[name = string("V_layer_cache_17_cast_fp16")]; + tensor x_131_axes_0 = const()[name = string("x_131_axes_0"), val = tensor([1])]; + tensor x_131_cast_fp16 = expand_dims(axes = x_131_axes_0, x = K_layer_cache_17_cast_fp16)[name = string("x_131_cast_fp16")]; + tensor var_5063 = const()[name = string("op_5063"), val = tensor([1, 2, 1, 1])]; + tensor x_133_cast_fp16 = tile(reps = var_5063, x = x_131_cast_fp16)[name = string("x_133_cast_fp16")]; + tensor var_5075 = const()[name = string("op_5075"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_35_cast_fp16 = reshape(shape = var_5075, x = x_133_cast_fp16)[name = string("key_states_35_cast_fp16")]; + tensor x_137_axes_0 = const()[name = string("x_137_axes_0"), val = tensor([1])]; + tensor x_137_cast_fp16 = expand_dims(axes = x_137_axes_0, x = V_layer_cache_17_cast_fp16)[name = string("x_137_cast_fp16")]; + tensor var_5083 = const()[name = string("op_5083"), val = tensor([1, 2, 1, 1])]; + tensor x_139_cast_fp16 = tile(reps = var_5083, x = x_137_cast_fp16)[name = string("x_139_cast_fp16")]; + tensor var_5095 = const()[name = string("op_5095"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_51_cast_fp16 = reshape(shape = var_5095, x = x_139_cast_fp16)[name = string("value_states_51_cast_fp16")]; + bool var_5110_transpose_x_1 = const()[name = string("op_5110_transpose_x_1"), val = bool(false)]; + bool var_5110_transpose_y_1 = const()[name = string("op_5110_transpose_y_1"), val = bool(true)]; + tensor var_5110 = matmul(transpose_x = var_5110_transpose_x_1, transpose_y = var_5110_transpose_y_1, x = query_states_33, y = key_states_35_cast_fp16)[name = string("op_5110")]; + fp16 var_5111_to_fp16 = const()[name = string("op_5111_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_49_cast_fp16 = mul(x = var_5110, y = var_5111_to_fp16)[name = string("attn_weights_49_cast_fp16")]; + tensor attn_weights_51_cast_fp16 = add(x = attn_weights_49_cast_fp16, y = causal_mask)[name = string("attn_weights_51_cast_fp16")]; + int32 var_5146 = const()[name = string("op_5146"), val = int32(-1)]; + tensor attn_weights_53_cast_fp16 = softmax(axis = var_5146, x = attn_weights_51_cast_fp16)[name = string("attn_weights_53_cast_fp16")]; + bool attn_output_81_transpose_x_0 = const()[name = string("attn_output_81_transpose_x_0"), val = bool(false)]; + bool attn_output_81_transpose_y_0 = const()[name = string("attn_output_81_transpose_y_0"), val = bool(false)]; + tensor attn_output_81_cast_fp16 = matmul(transpose_x = attn_output_81_transpose_x_0, transpose_y = attn_output_81_transpose_y_0, x = attn_weights_53_cast_fp16, y = value_states_51_cast_fp16)[name = string("attn_output_81_cast_fp16")]; + tensor var_5157_perm_0 = const()[name = string("op_5157_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_5161 = const()[name = string("op_5161"), val = tensor([1, 1, 2048])]; + tensor var_5157_cast_fp16 = transpose(perm = var_5157_perm_0, x = attn_output_81_cast_fp16)[name = string("transpose_34")]; + tensor attn_output_85_cast_fp16 = reshape(shape = var_5161, x = var_5157_cast_fp16)[name = string("attn_output_85_cast_fp16")]; + tensor var_5166 = const()[name = string("op_5166"), val = tensor([0, 2, 1])]; + string var_5182_pad_type_0 = const()[name = string("op_5182_pad_type_0"), val = string("valid")]; + int32 var_5182_groups_0 = const()[name = string("op_5182_groups_0"), val = int32(1)]; + tensor var_5182_strides_0 = const()[name = string("op_5182_strides_0"), val = tensor([1])]; + tensor var_5182_pad_0 = const()[name = string("op_5182_pad_0"), val = tensor([0, 0])]; + tensor var_5182_dilations_0 = const()[name = string("op_5182_dilations_0"), val = tensor([1])]; + tensor squeeze_8_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698177920))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702372288))))[name = string("squeeze_8_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_5167_cast_fp16 = transpose(perm = var_5166, x = attn_output_85_cast_fp16)[name = string("transpose_33")]; + tensor var_5182_cast_fp16 = conv(dilations = var_5182_dilations_0, groups = var_5182_groups_0, pad = var_5182_pad_0, pad_type = var_5182_pad_type_0, strides = var_5182_strides_0, weight = squeeze_8_cast_fp16_to_fp32_to_fp16_palettized, x = var_5167_cast_fp16)[name = string("op_5182_cast_fp16")]; + tensor var_5186 = const()[name = string("op_5186"), val = tensor([0, 2, 1])]; + tensor attn_output_89_cast_fp16 = transpose(perm = var_5186, x = var_5182_cast_fp16)[name = string("transpose_32")]; + tensor hidden_states_89_cast_fp16 = add(x = hidden_states_81_cast_fp16, y = attn_output_89_cast_fp16)[name = string("hidden_states_89_cast_fp16")]; + int32 var_5199 = const()[name = string("op_5199"), val = int32(-1)]; + fp16 const_266_promoted_to_fp16 = const()[name = string("const_266_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5201_cast_fp16 = mul(x = hidden_states_89_cast_fp16, y = const_266_promoted_to_fp16)[name = string("op_5201_cast_fp16")]; + bool input_155_interleave_0 = const()[name = string("input_155_interleave_0"), val = bool(false)]; + tensor input_155_cast_fp16 = concat(axis = var_5199, interleave = input_155_interleave_0, values = (hidden_states_89_cast_fp16, var_5201_cast_fp16))[name = string("input_155_cast_fp16")]; + tensor normed_141_axes_0 = const()[name = string("normed_141_axes_0"), val = tensor([-1])]; + fp16 var_5196_to_fp16 = const()[name = string("op_5196_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_141_cast_fp16 = layer_norm(axes = normed_141_axes_0, epsilon = var_5196_to_fp16, x = input_155_cast_fp16)[name = string("normed_141_cast_fp16")]; + tensor normed_143_begin_0 = const()[name = string("normed_143_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_143_end_0 = const()[name = string("normed_143_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_143_end_mask_0 = const()[name = string("normed_143_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_143_cast_fp16 = slice_by_index(begin = normed_143_begin_0, end = normed_143_end_0, end_mask = normed_143_end_mask_0, x = normed_141_cast_fp16)[name = string("normed_143_cast_fp16")]; + tensor const_269_promoted_to_fp16 = const()[name = string("const_269_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702503424)))]; + tensor x_141_cast_fp16 = mul(x = normed_143_cast_fp16, y = const_269_promoted_to_fp16)[name = string("x_141_cast_fp16")]; + tensor var_5226 = const()[name = string("op_5226"), val = tensor([0, 2, 1])]; + tensor input_157_axes_0 = const()[name = string("input_157_axes_0"), val = tensor([2])]; + tensor var_5227 = transpose(perm = var_5226, x = x_141_cast_fp16)[name = string("transpose_31")]; + tensor input_157 = expand_dims(axes = input_157_axes_0, x = var_5227)[name = string("input_157")]; + string input_159_pad_type_0 = const()[name = string("input_159_pad_type_0"), val = string("valid")]; + tensor input_159_strides_0 = const()[name = string("input_159_strides_0"), val = tensor([1, 1])]; + tensor input_159_pad_0 = const()[name = string("input_159_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_159_dilations_0 = const()[name = string("input_159_dilations_0"), val = tensor([1, 1])]; + int32 input_159_groups_0 = const()[name = string("input_159_groups_0"), val = int32(1)]; + tensor input_159 = conv(dilations = input_159_dilations_0, groups = input_159_groups_0, pad = input_159_pad_0, pad_type = input_159_pad_type_0, strides = input_159_strides_0, weight = model_model_layers_8_mlp_gate_proj_weight_palettized, x = input_157)[name = string("input_159")]; + string b_17_pad_type_0 = const()[name = string("b_17_pad_type_0"), val = string("valid")]; + tensor b_17_strides_0 = const()[name = string("b_17_strides_0"), val = tensor([1, 1])]; + tensor b_17_pad_0 = const()[name = string("b_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_17_dilations_0 = const()[name = string("b_17_dilations_0"), val = tensor([1, 1])]; + int32 b_17_groups_0 = const()[name = string("b_17_groups_0"), val = int32(1)]; + tensor b_17 = conv(dilations = b_17_dilations_0, groups = b_17_groups_0, pad = b_17_pad_0, pad_type = b_17_pad_type_0, strides = b_17_strides_0, weight = model_model_layers_8_mlp_up_proj_weight_palettized, x = input_157)[name = string("b_17")]; + tensor c_17 = silu(x = input_159)[name = string("c_17")]; + tensor input_161 = mul(x = c_17, y = b_17)[name = string("input_161")]; + string e_17_pad_type_0 = const()[name = string("e_17_pad_type_0"), val = string("valid")]; + tensor e_17_strides_0 = const()[name = string("e_17_strides_0"), val = tensor([1, 1])]; + tensor e_17_pad_0 = const()[name = string("e_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_17_dilations_0 = const()[name = string("e_17_dilations_0"), val = tensor([1, 1])]; + int32 e_17_groups_0 = const()[name = string("e_17_groups_0"), val = int32(1)]; + tensor e_17 = conv(dilations = e_17_dilations_0, groups = e_17_groups_0, pad = e_17_pad_0, pad_type = e_17_pad_type_0, strides = e_17_strides_0, weight = model_model_layers_8_mlp_down_proj_weight_palettized, x = input_161)[name = string("e_17")]; + tensor var_5249_axes_0 = const()[name = string("op_5249_axes_0"), val = tensor([2])]; + tensor var_5249 = squeeze(axes = var_5249_axes_0, x = e_17)[name = string("op_5249")]; + tensor var_5250 = const()[name = string("op_5250"), val = tensor([0, 2, 1])]; + tensor var_5251 = transpose(perm = var_5250, x = var_5249)[name = string("transpose_30")]; + tensor hidden_states_91_cast_fp16 = add(x = hidden_states_89_cast_fp16, y = var_5251)[name = string("hidden_states_91_cast_fp16")]; + int32 var_5263 = const()[name = string("op_5263"), val = int32(-1)]; + fp16 const_270_promoted_to_fp16 = const()[name = string("const_270_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5265_cast_fp16 = mul(x = hidden_states_91_cast_fp16, y = const_270_promoted_to_fp16)[name = string("op_5265_cast_fp16")]; + bool input_163_interleave_0 = const()[name = string("input_163_interleave_0"), val = bool(false)]; + tensor input_163_cast_fp16 = concat(axis = var_5263, interleave = input_163_interleave_0, values = (hidden_states_91_cast_fp16, var_5265_cast_fp16))[name = string("input_163_cast_fp16")]; + tensor normed_145_axes_0 = const()[name = string("normed_145_axes_0"), val = tensor([-1])]; + fp16 var_5260_to_fp16 = const()[name = string("op_5260_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_145_cast_fp16 = layer_norm(axes = normed_145_axes_0, epsilon = var_5260_to_fp16, x = input_163_cast_fp16)[name = string("normed_145_cast_fp16")]; + tensor normed_147_begin_0 = const()[name = string("normed_147_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_147_end_0 = const()[name = string("normed_147_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_147_end_mask_0 = const()[name = string("normed_147_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_147_cast_fp16 = slice_by_index(begin = normed_147_begin_0, end = normed_147_end_0, end_mask = normed_147_end_mask_0, x = normed_145_cast_fp16)[name = string("normed_147_cast_fp16")]; + tensor const_273_promoted_to_fp16 = const()[name = string("const_273_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702507584)))]; + tensor hidden_states_93_cast_fp16 = mul(x = normed_147_cast_fp16, y = const_273_promoted_to_fp16)[name = string("hidden_states_93_cast_fp16")]; + tensor var_5282 = const()[name = string("op_5282"), val = tensor([0, 2, 1])]; + tensor var_5285_axes_0 = const()[name = string("op_5285_axes_0"), val = tensor([2])]; + tensor var_5283_cast_fp16 = transpose(perm = var_5282, x = hidden_states_93_cast_fp16)[name = string("transpose_29")]; + tensor var_5285_cast_fp16 = expand_dims(axes = var_5285_axes_0, x = var_5283_cast_fp16)[name = string("op_5285_cast_fp16")]; + string var_5301_pad_type_0 = const()[name = string("op_5301_pad_type_0"), val = string("valid")]; + tensor var_5301_strides_0 = const()[name = string("op_5301_strides_0"), val = tensor([1, 1])]; + tensor var_5301_pad_0 = const()[name = string("op_5301_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5301_dilations_0 = const()[name = string("op_5301_dilations_0"), val = tensor([1, 1])]; + int32 var_5301_groups_0 = const()[name = string("op_5301_groups_0"), val = int32(1)]; + tensor var_5301 = conv(dilations = var_5301_dilations_0, groups = var_5301_groups_0, pad = var_5301_pad_0, pad_type = var_5301_pad_type_0, strides = var_5301_strides_0, weight = model_model_layers_9_self_attn_q_proj_weight_palettized, x = var_5285_cast_fp16)[name = string("op_5301")]; + tensor var_5306 = const()[name = string("op_5306"), val = tensor([1, 16, 1, 128])]; + tensor var_5307 = reshape(shape = var_5306, x = var_5301)[name = string("op_5307")]; + string var_5323_pad_type_0 = const()[name = string("op_5323_pad_type_0"), val = string("valid")]; + tensor var_5323_strides_0 = const()[name = string("op_5323_strides_0"), val = tensor([1, 1])]; + tensor var_5323_pad_0 = const()[name = string("op_5323_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5323_dilations_0 = const()[name = string("op_5323_dilations_0"), val = tensor([1, 1])]; + int32 var_5323_groups_0 = const()[name = string("op_5323_groups_0"), val = int32(1)]; + tensor var_5323 = conv(dilations = var_5323_dilations_0, groups = var_5323_groups_0, pad = var_5323_pad_0, pad_type = var_5323_pad_type_0, strides = var_5323_strides_0, weight = model_model_layers_9_self_attn_k_proj_weight_palettized, x = var_5285_cast_fp16)[name = string("op_5323")]; + tensor var_5328 = const()[name = string("op_5328"), val = tensor([1, 8, 1, 128])]; + tensor var_5329 = reshape(shape = var_5328, x = var_5323)[name = string("op_5329")]; + string var_5345_pad_type_0 = const()[name = string("op_5345_pad_type_0"), val = string("valid")]; + tensor var_5345_strides_0 = const()[name = string("op_5345_strides_0"), val = tensor([1, 1])]; + tensor var_5345_pad_0 = const()[name = string("op_5345_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5345_dilations_0 = const()[name = string("op_5345_dilations_0"), val = tensor([1, 1])]; + int32 var_5345_groups_0 = const()[name = string("op_5345_groups_0"), val = int32(1)]; + tensor var_5345 = conv(dilations = var_5345_dilations_0, groups = var_5345_groups_0, pad = var_5345_pad_0, pad_type = var_5345_pad_type_0, strides = var_5345_strides_0, weight = model_model_layers_9_self_attn_v_proj_weight_palettized, x = var_5285_cast_fp16)[name = string("op_5345")]; + tensor var_5350 = const()[name = string("op_5350"), val = tensor([1, 8, 1, 128])]; + tensor var_5351 = reshape(shape = var_5350, x = var_5345)[name = string("op_5351")]; + int32 var_5366 = const()[name = string("op_5366"), val = int32(-1)]; + fp16 const_274_promoted = const()[name = string("const_274_promoted"), val = fp16(-0x1p+0)]; + tensor var_5368 = mul(x = var_5307, y = const_274_promoted)[name = string("op_5368")]; + bool input_167_interleave_0 = const()[name = string("input_167_interleave_0"), val = bool(false)]; + tensor input_167 = concat(axis = var_5366, interleave = input_167_interleave_0, values = (var_5307, var_5368))[name = string("input_167")]; + tensor normed_149_axes_0 = const()[name = string("normed_149_axes_0"), val = tensor([-1])]; + fp16 var_5363_to_fp16 = const()[name = string("op_5363_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_149_cast_fp16 = layer_norm(axes = normed_149_axes_0, epsilon = var_5363_to_fp16, x = input_167)[name = string("normed_149_cast_fp16")]; + tensor normed_151_begin_0 = const()[name = string("normed_151_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_151_end_0 = const()[name = string("normed_151_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_151_end_mask_0 = const()[name = string("normed_151_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_151 = slice_by_index(begin = normed_151_begin_0, end = normed_151_end_0, end_mask = normed_151_end_mask_0, x = normed_149_cast_fp16)[name = string("normed_151")]; + tensor const_277 = const()[name = string("const_277"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702511744)))]; + tensor q_19 = mul(x = normed_151, y = const_277)[name = string("q_19")]; + int32 var_5391 = const()[name = string("op_5391"), val = int32(-1)]; + fp16 const_278_promoted = const()[name = string("const_278_promoted"), val = fp16(-0x1p+0)]; + tensor var_5393 = mul(x = var_5329, y = const_278_promoted)[name = string("op_5393")]; + bool input_169_interleave_0 = const()[name = string("input_169_interleave_0"), val = bool(false)]; + tensor input_169 = concat(axis = var_5391, interleave = input_169_interleave_0, values = (var_5329, var_5393))[name = string("input_169")]; + tensor normed_153_axes_0 = const()[name = string("normed_153_axes_0"), val = tensor([-1])]; + fp16 var_5388_to_fp16 = const()[name = string("op_5388_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_153_cast_fp16 = layer_norm(axes = normed_153_axes_0, epsilon = var_5388_to_fp16, x = input_169)[name = string("normed_153_cast_fp16")]; + tensor normed_155_begin_0 = const()[name = string("normed_155_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_155_end_0 = const()[name = string("normed_155_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_155_end_mask_0 = const()[name = string("normed_155_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_155 = slice_by_index(begin = normed_155_begin_0, end = normed_155_end_0, end_mask = normed_155_end_mask_0, x = normed_153_cast_fp16)[name = string("normed_155")]; + tensor const_281 = const()[name = string("const_281"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702512064)))]; + tensor k_19 = mul(x = normed_155, y = const_281)[name = string("k_19")]; + tensor var_5407 = mul(x = q_19, y = cos_1_cast_fp16)[name = string("op_5407")]; + tensor x1_37_begin_0 = const()[name = string("x1_37_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_37_end_0 = const()[name = string("x1_37_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_37_end_mask_0 = const()[name = string("x1_37_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_37 = slice_by_index(begin = x1_37_begin_0, end = x1_37_end_0, end_mask = x1_37_end_mask_0, x = q_19)[name = string("x1_37")]; + tensor x2_37_begin_0 = const()[name = string("x2_37_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_37_end_0 = const()[name = string("x2_37_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_37_end_mask_0 = const()[name = string("x2_37_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_37 = slice_by_index(begin = x2_37_begin_0, end = x2_37_end_0, end_mask = x2_37_end_mask_0, x = q_19)[name = string("x2_37")]; + fp16 const_284_promoted = const()[name = string("const_284_promoted"), val = fp16(-0x1p+0)]; + tensor var_5428 = mul(x = x2_37, y = const_284_promoted)[name = string("op_5428")]; + int32 var_5430 = const()[name = string("op_5430"), val = int32(-1)]; + bool var_5431_interleave_0 = const()[name = string("op_5431_interleave_0"), val = bool(false)]; + tensor var_5431 = concat(axis = var_5430, interleave = var_5431_interleave_0, values = (var_5428, x1_37))[name = string("op_5431")]; + tensor var_5432 = mul(x = var_5431, y = sin_1_cast_fp16)[name = string("op_5432")]; + tensor query_states_37 = add(x = var_5407, y = var_5432)[name = string("query_states_37")]; + tensor var_5435 = mul(x = k_19, y = cos_1_cast_fp16)[name = string("op_5435")]; + tensor x1_39_begin_0 = const()[name = string("x1_39_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_39_end_0 = const()[name = string("x1_39_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_39_end_mask_0 = const()[name = string("x1_39_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_39 = slice_by_index(begin = x1_39_begin_0, end = x1_39_end_0, end_mask = x1_39_end_mask_0, x = k_19)[name = string("x1_39")]; + tensor x2_39_begin_0 = const()[name = string("x2_39_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_39_end_0 = const()[name = string("x2_39_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_39_end_mask_0 = const()[name = string("x2_39_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_39 = slice_by_index(begin = x2_39_begin_0, end = x2_39_end_0, end_mask = x2_39_end_mask_0, x = k_19)[name = string("x2_39")]; + fp16 const_287_promoted = const()[name = string("const_287_promoted"), val = fp16(-0x1p+0)]; + tensor var_5456 = mul(x = x2_39, y = const_287_promoted)[name = string("op_5456")]; + int32 var_5458 = const()[name = string("op_5458"), val = int32(-1)]; + bool var_5459_interleave_0 = const()[name = string("op_5459_interleave_0"), val = bool(false)]; + tensor var_5459 = concat(axis = var_5458, interleave = var_5459_interleave_0, values = (var_5456, x1_39))[name = string("op_5459")]; + tensor var_5460 = mul(x = var_5459, y = sin_1_cast_fp16)[name = string("op_5460")]; + tensor key_states_37 = add(x = var_5435, y = var_5460)[name = string("key_states_37")]; + tensor expand_dims_108 = const()[name = string("expand_dims_108"), val = tensor([9])]; + tensor expand_dims_109 = const()[name = string("expand_dims_109"), val = tensor([0])]; + tensor expand_dims_111 = const()[name = string("expand_dims_111"), val = tensor([0])]; + tensor expand_dims_112 = const()[name = string("expand_dims_112"), val = tensor([10])]; + int32 concat_74_axis_0 = const()[name = string("concat_74_axis_0"), val = int32(0)]; + bool concat_74_interleave_0 = const()[name = string("concat_74_interleave_0"), val = bool(false)]; + tensor concat_74 = concat(axis = concat_74_axis_0, interleave = concat_74_interleave_0, values = (expand_dims_108, expand_dims_109, current_pos, expand_dims_111))[name = string("concat_74")]; + tensor concat_75_values1_0 = const()[name = string("concat_75_values1_0"), val = tensor([0])]; + tensor concat_75_values3_0 = const()[name = string("concat_75_values3_0"), val = tensor([0])]; + int32 concat_75_axis_0 = const()[name = string("concat_75_axis_0"), val = int32(0)]; + bool concat_75_interleave_0 = const()[name = string("concat_75_interleave_0"), val = bool(false)]; + tensor concat_75 = concat(axis = concat_75_axis_0, interleave = concat_75_interleave_0, values = (expand_dims_112, concat_75_values1_0, var_1001, concat_75_values3_0))[name = string("concat_75")]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_74, begin_mask = model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0, end = concat_75, end_mask = model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_19_stride_0, update = key_states_37, x = coreml_update_state_45)[name = string("model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_74_write_state")]; + tensor coreml_update_state_46 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_74")]; + tensor expand_dims_114 = const()[name = string("expand_dims_114"), val = tensor([37])]; + tensor expand_dims_115 = const()[name = string("expand_dims_115"), val = tensor([0])]; + tensor expand_dims_117 = const()[name = string("expand_dims_117"), val = tensor([0])]; + tensor expand_dims_118 = const()[name = string("expand_dims_118"), val = tensor([38])]; + int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; + bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; + tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_114, expand_dims_115, current_pos, expand_dims_117))[name = string("concat_78")]; + tensor concat_79_values1_0 = const()[name = string("concat_79_values1_0"), val = tensor([0])]; + tensor concat_79_values3_0 = const()[name = string("concat_79_values3_0"), val = tensor([0])]; + int32 concat_79_axis_0 = const()[name = string("concat_79_axis_0"), val = int32(0)]; + bool concat_79_interleave_0 = const()[name = string("concat_79_interleave_0"), val = bool(false)]; + tensor concat_79 = concat(axis = concat_79_axis_0, interleave = concat_79_interleave_0, values = (expand_dims_118, concat_79_values1_0, var_1001, concat_79_values3_0))[name = string("concat_79")]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_78, begin_mask = model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0, end = concat_79, end_mask = model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_20_stride_0, update = var_5351, x = coreml_update_state_46)[name = string("model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_75_write_state")]; + tensor coreml_update_state_47 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_75")]; + tensor var_5515_begin_0 = const()[name = string("op_5515_begin_0"), val = tensor([9, 0, 0, 0])]; + tensor var_5515_end_0 = const()[name = string("op_5515_end_0"), val = tensor([10, 8, 1024, 128])]; + tensor var_5515_end_mask_0 = const()[name = string("op_5515_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_5515_cast_fp16 = slice_by_index(begin = var_5515_begin_0, end = var_5515_end_0, end_mask = var_5515_end_mask_0, x = coreml_update_state_47)[name = string("op_5515_cast_fp16")]; + tensor K_layer_cache_19_axes_0 = const()[name = string("K_layer_cache_19_axes_0"), val = tensor([0])]; + tensor K_layer_cache_19_cast_fp16 = squeeze(axes = K_layer_cache_19_axes_0, x = var_5515_cast_fp16)[name = string("K_layer_cache_19_cast_fp16")]; + tensor var_5522_begin_0 = const()[name = string("op_5522_begin_0"), val = tensor([37, 0, 0, 0])]; + tensor var_5522_end_0 = const()[name = string("op_5522_end_0"), val = tensor([38, 8, 1024, 128])]; + tensor var_5522_end_mask_0 = const()[name = string("op_5522_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_5522_cast_fp16 = slice_by_index(begin = var_5522_begin_0, end = var_5522_end_0, end_mask = var_5522_end_mask_0, x = coreml_update_state_47)[name = string("op_5522_cast_fp16")]; + tensor V_layer_cache_19_axes_0 = const()[name = string("V_layer_cache_19_axes_0"), val = tensor([0])]; + tensor V_layer_cache_19_cast_fp16 = squeeze(axes = V_layer_cache_19_axes_0, x = var_5522_cast_fp16)[name = string("V_layer_cache_19_cast_fp16")]; + tensor x_147_axes_0 = const()[name = string("x_147_axes_0"), val = tensor([1])]; + tensor x_147_cast_fp16 = expand_dims(axes = x_147_axes_0, x = K_layer_cache_19_cast_fp16)[name = string("x_147_cast_fp16")]; + tensor var_5559 = const()[name = string("op_5559"), val = tensor([1, 2, 1, 1])]; + tensor x_149_cast_fp16 = tile(reps = var_5559, x = x_147_cast_fp16)[name = string("x_149_cast_fp16")]; + tensor var_5571 = const()[name = string("op_5571"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_39_cast_fp16 = reshape(shape = var_5571, x = x_149_cast_fp16)[name = string("key_states_39_cast_fp16")]; + tensor x_153_axes_0 = const()[name = string("x_153_axes_0"), val = tensor([1])]; + tensor x_153_cast_fp16 = expand_dims(axes = x_153_axes_0, x = V_layer_cache_19_cast_fp16)[name = string("x_153_cast_fp16")]; + tensor var_5579 = const()[name = string("op_5579"), val = tensor([1, 2, 1, 1])]; + tensor x_155_cast_fp16 = tile(reps = var_5579, x = x_153_cast_fp16)[name = string("x_155_cast_fp16")]; + tensor var_5591 = const()[name = string("op_5591"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_57_cast_fp16 = reshape(shape = var_5591, x = x_155_cast_fp16)[name = string("value_states_57_cast_fp16")]; + bool var_5606_transpose_x_1 = const()[name = string("op_5606_transpose_x_1"), val = bool(false)]; + bool var_5606_transpose_y_1 = const()[name = string("op_5606_transpose_y_1"), val = bool(true)]; + tensor var_5606 = matmul(transpose_x = var_5606_transpose_x_1, transpose_y = var_5606_transpose_y_1, x = query_states_37, y = key_states_39_cast_fp16)[name = string("op_5606")]; + fp16 var_5607_to_fp16 = const()[name = string("op_5607_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_55_cast_fp16 = mul(x = var_5606, y = var_5607_to_fp16)[name = string("attn_weights_55_cast_fp16")]; + tensor attn_weights_57_cast_fp16 = add(x = attn_weights_55_cast_fp16, y = causal_mask)[name = string("attn_weights_57_cast_fp16")]; + int32 var_5642 = const()[name = string("op_5642"), val = int32(-1)]; + tensor attn_weights_59_cast_fp16 = softmax(axis = var_5642, x = attn_weights_57_cast_fp16)[name = string("attn_weights_59_cast_fp16")]; + bool attn_output_91_transpose_x_0 = const()[name = string("attn_output_91_transpose_x_0"), val = bool(false)]; + bool attn_output_91_transpose_y_0 = const()[name = string("attn_output_91_transpose_y_0"), val = bool(false)]; + tensor attn_output_91_cast_fp16 = matmul(transpose_x = attn_output_91_transpose_x_0, transpose_y = attn_output_91_transpose_y_0, x = attn_weights_59_cast_fp16, y = value_states_57_cast_fp16)[name = string("attn_output_91_cast_fp16")]; + tensor var_5653_perm_0 = const()[name = string("op_5653_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_5657 = const()[name = string("op_5657"), val = tensor([1, 1, 2048])]; + tensor var_5653_cast_fp16 = transpose(perm = var_5653_perm_0, x = attn_output_91_cast_fp16)[name = string("transpose_28")]; + tensor attn_output_95_cast_fp16 = reshape(shape = var_5657, x = var_5653_cast_fp16)[name = string("attn_output_95_cast_fp16")]; + tensor var_5662 = const()[name = string("op_5662"), val = tensor([0, 2, 1])]; + string var_5678_pad_type_0 = const()[name = string("op_5678_pad_type_0"), val = string("valid")]; + int32 var_5678_groups_0 = const()[name = string("op_5678_groups_0"), val = int32(1)]; + tensor var_5678_strides_0 = const()[name = string("op_5678_strides_0"), val = tensor([1])]; + tensor var_5678_pad_0 = const()[name = string("op_5678_pad_0"), val = tensor([0, 0])]; + tensor var_5678_dilations_0 = const()[name = string("op_5678_dilations_0"), val = tensor([1])]; + tensor squeeze_9_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702512384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706706752))))[name = string("squeeze_9_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_5663_cast_fp16 = transpose(perm = var_5662, x = attn_output_95_cast_fp16)[name = string("transpose_27")]; + tensor var_5678_cast_fp16 = conv(dilations = var_5678_dilations_0, groups = var_5678_groups_0, pad = var_5678_pad_0, pad_type = var_5678_pad_type_0, strides = var_5678_strides_0, weight = squeeze_9_cast_fp16_to_fp32_to_fp16_palettized, x = var_5663_cast_fp16)[name = string("op_5678_cast_fp16")]; + tensor var_5682 = const()[name = string("op_5682"), val = tensor([0, 2, 1])]; + tensor attn_output_99_cast_fp16 = transpose(perm = var_5682, x = var_5678_cast_fp16)[name = string("transpose_26")]; + tensor hidden_states_99_cast_fp16 = add(x = hidden_states_91_cast_fp16, y = attn_output_99_cast_fp16)[name = string("hidden_states_99_cast_fp16")]; + int32 var_5695 = const()[name = string("op_5695"), val = int32(-1)]; + fp16 const_296_promoted_to_fp16 = const()[name = string("const_296_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5697_cast_fp16 = mul(x = hidden_states_99_cast_fp16, y = const_296_promoted_to_fp16)[name = string("op_5697_cast_fp16")]; + bool input_173_interleave_0 = const()[name = string("input_173_interleave_0"), val = bool(false)]; + tensor input_173_cast_fp16 = concat(axis = var_5695, interleave = input_173_interleave_0, values = (hidden_states_99_cast_fp16, var_5697_cast_fp16))[name = string("input_173_cast_fp16")]; + tensor normed_157_axes_0 = const()[name = string("normed_157_axes_0"), val = tensor([-1])]; + fp16 var_5692_to_fp16 = const()[name = string("op_5692_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_157_cast_fp16 = layer_norm(axes = normed_157_axes_0, epsilon = var_5692_to_fp16, x = input_173_cast_fp16)[name = string("normed_157_cast_fp16")]; + tensor normed_159_begin_0 = const()[name = string("normed_159_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_159_end_0 = const()[name = string("normed_159_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_159_end_mask_0 = const()[name = string("normed_159_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_159_cast_fp16 = slice_by_index(begin = normed_159_begin_0, end = normed_159_end_0, end_mask = normed_159_end_mask_0, x = normed_157_cast_fp16)[name = string("normed_159_cast_fp16")]; + tensor const_299_promoted_to_fp16 = const()[name = string("const_299_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706837888)))]; + tensor x_157_cast_fp16 = mul(x = normed_159_cast_fp16, y = const_299_promoted_to_fp16)[name = string("x_157_cast_fp16")]; + tensor var_5722 = const()[name = string("op_5722"), val = tensor([0, 2, 1])]; + tensor input_175_axes_0 = const()[name = string("input_175_axes_0"), val = tensor([2])]; + tensor var_5723 = transpose(perm = var_5722, x = x_157_cast_fp16)[name = string("transpose_25")]; + tensor input_175 = expand_dims(axes = input_175_axes_0, x = var_5723)[name = string("input_175")]; + string input_177_pad_type_0 = const()[name = string("input_177_pad_type_0"), val = string("valid")]; + tensor input_177_strides_0 = const()[name = string("input_177_strides_0"), val = tensor([1, 1])]; + tensor input_177_pad_0 = const()[name = string("input_177_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_177_dilations_0 = const()[name = string("input_177_dilations_0"), val = tensor([1, 1])]; + int32 input_177_groups_0 = const()[name = string("input_177_groups_0"), val = int32(1)]; + tensor input_177 = conv(dilations = input_177_dilations_0, groups = input_177_groups_0, pad = input_177_pad_0, pad_type = input_177_pad_type_0, strides = input_177_strides_0, weight = model_model_layers_9_mlp_gate_proj_weight_palettized, x = input_175)[name = string("input_177")]; + string b_19_pad_type_0 = const()[name = string("b_19_pad_type_0"), val = string("valid")]; + tensor b_19_strides_0 = const()[name = string("b_19_strides_0"), val = tensor([1, 1])]; + tensor b_19_pad_0 = const()[name = string("b_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_19_dilations_0 = const()[name = string("b_19_dilations_0"), val = tensor([1, 1])]; + int32 b_19_groups_0 = const()[name = string("b_19_groups_0"), val = int32(1)]; + tensor b_19 = conv(dilations = b_19_dilations_0, groups = b_19_groups_0, pad = b_19_pad_0, pad_type = b_19_pad_type_0, strides = b_19_strides_0, weight = model_model_layers_9_mlp_up_proj_weight_palettized, x = input_175)[name = string("b_19")]; + tensor c_19 = silu(x = input_177)[name = string("c_19")]; + tensor input_179 = mul(x = c_19, y = b_19)[name = string("input_179")]; + string e_19_pad_type_0 = const()[name = string("e_19_pad_type_0"), val = string("valid")]; + tensor e_19_strides_0 = const()[name = string("e_19_strides_0"), val = tensor([1, 1])]; + tensor e_19_pad_0 = const()[name = string("e_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_19_dilations_0 = const()[name = string("e_19_dilations_0"), val = tensor([1, 1])]; + int32 e_19_groups_0 = const()[name = string("e_19_groups_0"), val = int32(1)]; + tensor e_19 = conv(dilations = e_19_dilations_0, groups = e_19_groups_0, pad = e_19_pad_0, pad_type = e_19_pad_type_0, strides = e_19_strides_0, weight = model_model_layers_9_mlp_down_proj_weight_palettized, x = input_179)[name = string("e_19")]; + tensor var_5745_axes_0 = const()[name = string("op_5745_axes_0"), val = tensor([2])]; + tensor var_5745 = squeeze(axes = var_5745_axes_0, x = e_19)[name = string("op_5745")]; + tensor var_5746 = const()[name = string("op_5746"), val = tensor([0, 2, 1])]; + tensor var_5747 = transpose(perm = var_5746, x = var_5745)[name = string("transpose_24")]; + tensor hidden_states_101_cast_fp16 = add(x = hidden_states_99_cast_fp16, y = var_5747)[name = string("hidden_states_101_cast_fp16")]; + int32 var_5759 = const()[name = string("op_5759"), val = int32(-1)]; + fp16 const_300_promoted_to_fp16 = const()[name = string("const_300_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5761_cast_fp16 = mul(x = hidden_states_101_cast_fp16, y = const_300_promoted_to_fp16)[name = string("op_5761_cast_fp16")]; + bool input_181_interleave_0 = const()[name = string("input_181_interleave_0"), val = bool(false)]; + tensor input_181_cast_fp16 = concat(axis = var_5759, interleave = input_181_interleave_0, values = (hidden_states_101_cast_fp16, var_5761_cast_fp16))[name = string("input_181_cast_fp16")]; + tensor normed_161_axes_0 = const()[name = string("normed_161_axes_0"), val = tensor([-1])]; + fp16 var_5756_to_fp16 = const()[name = string("op_5756_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_161_cast_fp16 = layer_norm(axes = normed_161_axes_0, epsilon = var_5756_to_fp16, x = input_181_cast_fp16)[name = string("normed_161_cast_fp16")]; + tensor normed_163_begin_0 = const()[name = string("normed_163_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_163_end_0 = const()[name = string("normed_163_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_163_end_mask_0 = const()[name = string("normed_163_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_163_cast_fp16 = slice_by_index(begin = normed_163_begin_0, end = normed_163_end_0, end_mask = normed_163_end_mask_0, x = normed_161_cast_fp16)[name = string("normed_163_cast_fp16")]; + tensor const_303_promoted_to_fp16 = const()[name = string("const_303_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706842048)))]; + tensor hidden_states_103_cast_fp16 = mul(x = normed_163_cast_fp16, y = const_303_promoted_to_fp16)[name = string("hidden_states_103_cast_fp16")]; + tensor var_5778 = const()[name = string("op_5778"), val = tensor([0, 2, 1])]; + tensor var_5781_axes_0 = const()[name = string("op_5781_axes_0"), val = tensor([2])]; + tensor var_5779_cast_fp16 = transpose(perm = var_5778, x = hidden_states_103_cast_fp16)[name = string("transpose_23")]; + tensor var_5781_cast_fp16 = expand_dims(axes = var_5781_axes_0, x = var_5779_cast_fp16)[name = string("op_5781_cast_fp16")]; + string var_5797_pad_type_0 = const()[name = string("op_5797_pad_type_0"), val = string("valid")]; + tensor var_5797_strides_0 = const()[name = string("op_5797_strides_0"), val = tensor([1, 1])]; + tensor var_5797_pad_0 = const()[name = string("op_5797_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5797_dilations_0 = const()[name = string("op_5797_dilations_0"), val = tensor([1, 1])]; + int32 var_5797_groups_0 = const()[name = string("op_5797_groups_0"), val = int32(1)]; + tensor var_5797 = conv(dilations = var_5797_dilations_0, groups = var_5797_groups_0, pad = var_5797_pad_0, pad_type = var_5797_pad_type_0, strides = var_5797_strides_0, weight = model_model_layers_10_self_attn_q_proj_weight_palettized, x = var_5781_cast_fp16)[name = string("op_5797")]; + tensor var_5802 = const()[name = string("op_5802"), val = tensor([1, 16, 1, 128])]; + tensor var_5803 = reshape(shape = var_5802, x = var_5797)[name = string("op_5803")]; + string var_5819_pad_type_0 = const()[name = string("op_5819_pad_type_0"), val = string("valid")]; + tensor var_5819_strides_0 = const()[name = string("op_5819_strides_0"), val = tensor([1, 1])]; + tensor var_5819_pad_0 = const()[name = string("op_5819_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5819_dilations_0 = const()[name = string("op_5819_dilations_0"), val = tensor([1, 1])]; + int32 var_5819_groups_0 = const()[name = string("op_5819_groups_0"), val = int32(1)]; + tensor var_5819 = conv(dilations = var_5819_dilations_0, groups = var_5819_groups_0, pad = var_5819_pad_0, pad_type = var_5819_pad_type_0, strides = var_5819_strides_0, weight = model_model_layers_10_self_attn_k_proj_weight_palettized, x = var_5781_cast_fp16)[name = string("op_5819")]; + tensor var_5824 = const()[name = string("op_5824"), val = tensor([1, 8, 1, 128])]; + tensor var_5825 = reshape(shape = var_5824, x = var_5819)[name = string("op_5825")]; + string var_5841_pad_type_0 = const()[name = string("op_5841_pad_type_0"), val = string("valid")]; + tensor var_5841_strides_0 = const()[name = string("op_5841_strides_0"), val = tensor([1, 1])]; + tensor var_5841_pad_0 = const()[name = string("op_5841_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5841_dilations_0 = const()[name = string("op_5841_dilations_0"), val = tensor([1, 1])]; + int32 var_5841_groups_0 = const()[name = string("op_5841_groups_0"), val = int32(1)]; + tensor var_5841 = conv(dilations = var_5841_dilations_0, groups = var_5841_groups_0, pad = var_5841_pad_0, pad_type = var_5841_pad_type_0, strides = var_5841_strides_0, weight = model_model_layers_10_self_attn_v_proj_weight_palettized, x = var_5781_cast_fp16)[name = string("op_5841")]; + tensor var_5846 = const()[name = string("op_5846"), val = tensor([1, 8, 1, 128])]; + tensor var_5847 = reshape(shape = var_5846, x = var_5841)[name = string("op_5847")]; + int32 var_5862 = const()[name = string("op_5862"), val = int32(-1)]; + fp16 const_304_promoted = const()[name = string("const_304_promoted"), val = fp16(-0x1p+0)]; + tensor var_5864 = mul(x = var_5803, y = const_304_promoted)[name = string("op_5864")]; + bool input_185_interleave_0 = const()[name = string("input_185_interleave_0"), val = bool(false)]; + tensor input_185 = concat(axis = var_5862, interleave = input_185_interleave_0, values = (var_5803, var_5864))[name = string("input_185")]; + tensor normed_165_axes_0 = const()[name = string("normed_165_axes_0"), val = tensor([-1])]; + fp16 var_5859_to_fp16 = const()[name = string("op_5859_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_165_cast_fp16 = layer_norm(axes = normed_165_axes_0, epsilon = var_5859_to_fp16, x = input_185)[name = string("normed_165_cast_fp16")]; + tensor normed_167_begin_0 = const()[name = string("normed_167_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_167_end_0 = const()[name = string("normed_167_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_167_end_mask_0 = const()[name = string("normed_167_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_167 = slice_by_index(begin = normed_167_begin_0, end = normed_167_end_0, end_mask = normed_167_end_mask_0, x = normed_165_cast_fp16)[name = string("normed_167")]; + tensor const_307 = const()[name = string("const_307"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706846208)))]; + tensor q_21 = mul(x = normed_167, y = const_307)[name = string("q_21")]; + int32 var_5887 = const()[name = string("op_5887"), val = int32(-1)]; + fp16 const_308_promoted = const()[name = string("const_308_promoted"), val = fp16(-0x1p+0)]; + tensor var_5889 = mul(x = var_5825, y = const_308_promoted)[name = string("op_5889")]; + bool input_187_interleave_0 = const()[name = string("input_187_interleave_0"), val = bool(false)]; + tensor input_187 = concat(axis = var_5887, interleave = input_187_interleave_0, values = (var_5825, var_5889))[name = string("input_187")]; + tensor normed_169_axes_0 = const()[name = string("normed_169_axes_0"), val = tensor([-1])]; + fp16 var_5884_to_fp16 = const()[name = string("op_5884_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_169_cast_fp16 = layer_norm(axes = normed_169_axes_0, epsilon = var_5884_to_fp16, x = input_187)[name = string("normed_169_cast_fp16")]; + tensor normed_171_begin_0 = const()[name = string("normed_171_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_171_end_0 = const()[name = string("normed_171_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_171_end_mask_0 = const()[name = string("normed_171_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_171 = slice_by_index(begin = normed_171_begin_0, end = normed_171_end_0, end_mask = normed_171_end_mask_0, x = normed_169_cast_fp16)[name = string("normed_171")]; + tensor const_311 = const()[name = string("const_311"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706846528)))]; + tensor k_21 = mul(x = normed_171, y = const_311)[name = string("k_21")]; + tensor var_5903 = mul(x = q_21, y = cos_1_cast_fp16)[name = string("op_5903")]; + tensor x1_41_begin_0 = const()[name = string("x1_41_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_41_end_0 = const()[name = string("x1_41_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_41_end_mask_0 = const()[name = string("x1_41_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_41 = slice_by_index(begin = x1_41_begin_0, end = x1_41_end_0, end_mask = x1_41_end_mask_0, x = q_21)[name = string("x1_41")]; + tensor x2_41_begin_0 = const()[name = string("x2_41_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_41_end_0 = const()[name = string("x2_41_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_41_end_mask_0 = const()[name = string("x2_41_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_41 = slice_by_index(begin = x2_41_begin_0, end = x2_41_end_0, end_mask = x2_41_end_mask_0, x = q_21)[name = string("x2_41")]; + fp16 const_314_promoted = const()[name = string("const_314_promoted"), val = fp16(-0x1p+0)]; + tensor var_5924 = mul(x = x2_41, y = const_314_promoted)[name = string("op_5924")]; + int32 var_5926 = const()[name = string("op_5926"), val = int32(-1)]; + bool var_5927_interleave_0 = const()[name = string("op_5927_interleave_0"), val = bool(false)]; + tensor var_5927 = concat(axis = var_5926, interleave = var_5927_interleave_0, values = (var_5924, x1_41))[name = string("op_5927")]; + tensor var_5928 = mul(x = var_5927, y = sin_1_cast_fp16)[name = string("op_5928")]; + tensor query_states_41 = add(x = var_5903, y = var_5928)[name = string("query_states_41")]; + tensor var_5931 = mul(x = k_21, y = cos_1_cast_fp16)[name = string("op_5931")]; + tensor x1_43_begin_0 = const()[name = string("x1_43_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_43_end_0 = const()[name = string("x1_43_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_43_end_mask_0 = const()[name = string("x1_43_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_43 = slice_by_index(begin = x1_43_begin_0, end = x1_43_end_0, end_mask = x1_43_end_mask_0, x = k_21)[name = string("x1_43")]; + tensor x2_43_begin_0 = const()[name = string("x2_43_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_43_end_0 = const()[name = string("x2_43_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_43_end_mask_0 = const()[name = string("x2_43_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_43 = slice_by_index(begin = x2_43_begin_0, end = x2_43_end_0, end_mask = x2_43_end_mask_0, x = k_21)[name = string("x2_43")]; + fp16 const_317_promoted = const()[name = string("const_317_promoted"), val = fp16(-0x1p+0)]; + tensor var_5952 = mul(x = x2_43, y = const_317_promoted)[name = string("op_5952")]; + int32 var_5954 = const()[name = string("op_5954"), val = int32(-1)]; + bool var_5955_interleave_0 = const()[name = string("op_5955_interleave_0"), val = bool(false)]; + tensor var_5955 = concat(axis = var_5954, interleave = var_5955_interleave_0, values = (var_5952, x1_43))[name = string("op_5955")]; + tensor var_5956 = mul(x = var_5955, y = sin_1_cast_fp16)[name = string("op_5956")]; + tensor key_states_41 = add(x = var_5931, y = var_5956)[name = string("key_states_41")]; + tensor expand_dims_120 = const()[name = string("expand_dims_120"), val = tensor([10])]; + tensor expand_dims_121 = const()[name = string("expand_dims_121"), val = tensor([0])]; + tensor expand_dims_123 = const()[name = string("expand_dims_123"), val = tensor([0])]; + tensor expand_dims_124 = const()[name = string("expand_dims_124"), val = tensor([11])]; + int32 concat_82_axis_0 = const()[name = string("concat_82_axis_0"), val = int32(0)]; + bool concat_82_interleave_0 = const()[name = string("concat_82_interleave_0"), val = bool(false)]; + tensor concat_82 = concat(axis = concat_82_axis_0, interleave = concat_82_interleave_0, values = (expand_dims_120, expand_dims_121, current_pos, expand_dims_123))[name = string("concat_82")]; + tensor concat_83_values1_0 = const()[name = string("concat_83_values1_0"), val = tensor([0])]; + tensor concat_83_values3_0 = const()[name = string("concat_83_values3_0"), val = tensor([0])]; + int32 concat_83_axis_0 = const()[name = string("concat_83_axis_0"), val = int32(0)]; + bool concat_83_interleave_0 = const()[name = string("concat_83_interleave_0"), val = bool(false)]; + tensor concat_83 = concat(axis = concat_83_axis_0, interleave = concat_83_interleave_0, values = (expand_dims_124, concat_83_values1_0, var_1001, concat_83_values3_0))[name = string("concat_83")]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_82, begin_mask = model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0, end = concat_83, end_mask = model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_21_stride_0, update = key_states_41, x = coreml_update_state_47)[name = string("model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_76_write_state")]; + tensor coreml_update_state_48 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_76")]; + tensor expand_dims_126 = const()[name = string("expand_dims_126"), val = tensor([38])]; + tensor expand_dims_127 = const()[name = string("expand_dims_127"), val = tensor([0])]; + tensor expand_dims_129 = const()[name = string("expand_dims_129"), val = tensor([0])]; + tensor expand_dims_130 = const()[name = string("expand_dims_130"), val = tensor([39])]; + int32 concat_86_axis_0 = const()[name = string("concat_86_axis_0"), val = int32(0)]; + bool concat_86_interleave_0 = const()[name = string("concat_86_interleave_0"), val = bool(false)]; + tensor concat_86 = concat(axis = concat_86_axis_0, interleave = concat_86_interleave_0, values = (expand_dims_126, expand_dims_127, current_pos, expand_dims_129))[name = string("concat_86")]; + tensor concat_87_values1_0 = const()[name = string("concat_87_values1_0"), val = tensor([0])]; + tensor concat_87_values3_0 = const()[name = string("concat_87_values3_0"), val = tensor([0])]; + int32 concat_87_axis_0 = const()[name = string("concat_87_axis_0"), val = int32(0)]; + bool concat_87_interleave_0 = const()[name = string("concat_87_interleave_0"), val = bool(false)]; + tensor concat_87 = concat(axis = concat_87_axis_0, interleave = concat_87_interleave_0, values = (expand_dims_130, concat_87_values1_0, var_1001, concat_87_values3_0))[name = string("concat_87")]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_86, begin_mask = model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0, end = concat_87, end_mask = model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_22_stride_0, update = var_5847, x = coreml_update_state_48)[name = string("model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_77_write_state")]; + tensor coreml_update_state_49 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_77")]; + tensor var_6011_begin_0 = const()[name = string("op_6011_begin_0"), val = tensor([10, 0, 0, 0])]; + tensor var_6011_end_0 = const()[name = string("op_6011_end_0"), val = tensor([11, 8, 1024, 128])]; + tensor var_6011_end_mask_0 = const()[name = string("op_6011_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_6011_cast_fp16 = slice_by_index(begin = var_6011_begin_0, end = var_6011_end_0, end_mask = var_6011_end_mask_0, x = coreml_update_state_49)[name = string("op_6011_cast_fp16")]; + tensor K_layer_cache_21_axes_0 = const()[name = string("K_layer_cache_21_axes_0"), val = tensor([0])]; + tensor K_layer_cache_21_cast_fp16 = squeeze(axes = K_layer_cache_21_axes_0, x = var_6011_cast_fp16)[name = string("K_layer_cache_21_cast_fp16")]; + tensor var_6018_begin_0 = const()[name = string("op_6018_begin_0"), val = tensor([38, 0, 0, 0])]; + tensor var_6018_end_0 = const()[name = string("op_6018_end_0"), val = tensor([39, 8, 1024, 128])]; + tensor var_6018_end_mask_0 = const()[name = string("op_6018_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_6018_cast_fp16 = slice_by_index(begin = var_6018_begin_0, end = var_6018_end_0, end_mask = var_6018_end_mask_0, x = coreml_update_state_49)[name = string("op_6018_cast_fp16")]; + tensor V_layer_cache_21_axes_0 = const()[name = string("V_layer_cache_21_axes_0"), val = tensor([0])]; + tensor V_layer_cache_21_cast_fp16 = squeeze(axes = V_layer_cache_21_axes_0, x = var_6018_cast_fp16)[name = string("V_layer_cache_21_cast_fp16")]; + tensor x_163_axes_0 = const()[name = string("x_163_axes_0"), val = tensor([1])]; + tensor x_163_cast_fp16 = expand_dims(axes = x_163_axes_0, x = K_layer_cache_21_cast_fp16)[name = string("x_163_cast_fp16")]; + tensor var_6055 = const()[name = string("op_6055"), val = tensor([1, 2, 1, 1])]; + tensor x_165_cast_fp16 = tile(reps = var_6055, x = x_163_cast_fp16)[name = string("x_165_cast_fp16")]; + tensor var_6067 = const()[name = string("op_6067"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_43_cast_fp16 = reshape(shape = var_6067, x = x_165_cast_fp16)[name = string("key_states_43_cast_fp16")]; + tensor x_169_axes_0 = const()[name = string("x_169_axes_0"), val = tensor([1])]; + tensor x_169_cast_fp16 = expand_dims(axes = x_169_axes_0, x = V_layer_cache_21_cast_fp16)[name = string("x_169_cast_fp16")]; + tensor var_6075 = const()[name = string("op_6075"), val = tensor([1, 2, 1, 1])]; + tensor x_171_cast_fp16 = tile(reps = var_6075, x = x_169_cast_fp16)[name = string("x_171_cast_fp16")]; + tensor var_6087 = const()[name = string("op_6087"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_63_cast_fp16 = reshape(shape = var_6087, x = x_171_cast_fp16)[name = string("value_states_63_cast_fp16")]; + bool var_6102_transpose_x_1 = const()[name = string("op_6102_transpose_x_1"), val = bool(false)]; + bool var_6102_transpose_y_1 = const()[name = string("op_6102_transpose_y_1"), val = bool(true)]; + tensor var_6102 = matmul(transpose_x = var_6102_transpose_x_1, transpose_y = var_6102_transpose_y_1, x = query_states_41, y = key_states_43_cast_fp16)[name = string("op_6102")]; + fp16 var_6103_to_fp16 = const()[name = string("op_6103_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_61_cast_fp16 = mul(x = var_6102, y = var_6103_to_fp16)[name = string("attn_weights_61_cast_fp16")]; + tensor attn_weights_63_cast_fp16 = add(x = attn_weights_61_cast_fp16, y = causal_mask)[name = string("attn_weights_63_cast_fp16")]; + int32 var_6138 = const()[name = string("op_6138"), val = int32(-1)]; + tensor attn_weights_65_cast_fp16 = softmax(axis = var_6138, x = attn_weights_63_cast_fp16)[name = string("attn_weights_65_cast_fp16")]; + bool attn_output_101_transpose_x_0 = const()[name = string("attn_output_101_transpose_x_0"), val = bool(false)]; + bool attn_output_101_transpose_y_0 = const()[name = string("attn_output_101_transpose_y_0"), val = bool(false)]; + tensor attn_output_101_cast_fp16 = matmul(transpose_x = attn_output_101_transpose_x_0, transpose_y = attn_output_101_transpose_y_0, x = attn_weights_65_cast_fp16, y = value_states_63_cast_fp16)[name = string("attn_output_101_cast_fp16")]; + tensor var_6149_perm_0 = const()[name = string("op_6149_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_6153 = const()[name = string("op_6153"), val = tensor([1, 1, 2048])]; + tensor var_6149_cast_fp16 = transpose(perm = var_6149_perm_0, x = attn_output_101_cast_fp16)[name = string("transpose_22")]; + tensor attn_output_105_cast_fp16 = reshape(shape = var_6153, x = var_6149_cast_fp16)[name = string("attn_output_105_cast_fp16")]; + tensor var_6158 = const()[name = string("op_6158"), val = tensor([0, 2, 1])]; + string var_6174_pad_type_0 = const()[name = string("op_6174_pad_type_0"), val = string("valid")]; + int32 var_6174_groups_0 = const()[name = string("op_6174_groups_0"), val = int32(1)]; + tensor var_6174_strides_0 = const()[name = string("op_6174_strides_0"), val = tensor([1])]; + tensor var_6174_pad_0 = const()[name = string("op_6174_pad_0"), val = tensor([0, 0])]; + tensor var_6174_dilations_0 = const()[name = string("op_6174_dilations_0"), val = tensor([1])]; + tensor squeeze_10_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706846848))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711041216))))[name = string("squeeze_10_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_6159_cast_fp16 = transpose(perm = var_6158, x = attn_output_105_cast_fp16)[name = string("transpose_21")]; + tensor var_6174_cast_fp16 = conv(dilations = var_6174_dilations_0, groups = var_6174_groups_0, pad = var_6174_pad_0, pad_type = var_6174_pad_type_0, strides = var_6174_strides_0, weight = squeeze_10_cast_fp16_to_fp32_to_fp16_palettized, x = var_6159_cast_fp16)[name = string("op_6174_cast_fp16")]; + tensor var_6178 = const()[name = string("op_6178"), val = tensor([0, 2, 1])]; + tensor attn_output_109_cast_fp16 = transpose(perm = var_6178, x = var_6174_cast_fp16)[name = string("transpose_20")]; + tensor hidden_states_109_cast_fp16 = add(x = hidden_states_101_cast_fp16, y = attn_output_109_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; + int32 var_6191 = const()[name = string("op_6191"), val = int32(-1)]; + fp16 const_326_promoted_to_fp16 = const()[name = string("const_326_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6193_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = const_326_promoted_to_fp16)[name = string("op_6193_cast_fp16")]; + bool input_191_interleave_0 = const()[name = string("input_191_interleave_0"), val = bool(false)]; + tensor input_191_cast_fp16 = concat(axis = var_6191, interleave = input_191_interleave_0, values = (hidden_states_109_cast_fp16, var_6193_cast_fp16))[name = string("input_191_cast_fp16")]; + tensor normed_173_axes_0 = const()[name = string("normed_173_axes_0"), val = tensor([-1])]; + fp16 var_6188_to_fp16 = const()[name = string("op_6188_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_173_cast_fp16 = layer_norm(axes = normed_173_axes_0, epsilon = var_6188_to_fp16, x = input_191_cast_fp16)[name = string("normed_173_cast_fp16")]; + tensor normed_175_begin_0 = const()[name = string("normed_175_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_175_end_0 = const()[name = string("normed_175_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_175_end_mask_0 = const()[name = string("normed_175_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_175_cast_fp16 = slice_by_index(begin = normed_175_begin_0, end = normed_175_end_0, end_mask = normed_175_end_mask_0, x = normed_173_cast_fp16)[name = string("normed_175_cast_fp16")]; + tensor const_329_promoted_to_fp16 = const()[name = string("const_329_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711172352)))]; + tensor x_173_cast_fp16 = mul(x = normed_175_cast_fp16, y = const_329_promoted_to_fp16)[name = string("x_173_cast_fp16")]; + tensor var_6218 = const()[name = string("op_6218"), val = tensor([0, 2, 1])]; + tensor input_193_axes_0 = const()[name = string("input_193_axes_0"), val = tensor([2])]; + tensor var_6219 = transpose(perm = var_6218, x = x_173_cast_fp16)[name = string("transpose_19")]; + tensor input_193 = expand_dims(axes = input_193_axes_0, x = var_6219)[name = string("input_193")]; + string input_195_pad_type_0 = const()[name = string("input_195_pad_type_0"), val = string("valid")]; + tensor input_195_strides_0 = const()[name = string("input_195_strides_0"), val = tensor([1, 1])]; + tensor input_195_pad_0 = const()[name = string("input_195_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_195_dilations_0 = const()[name = string("input_195_dilations_0"), val = tensor([1, 1])]; + int32 input_195_groups_0 = const()[name = string("input_195_groups_0"), val = int32(1)]; + tensor input_195 = conv(dilations = input_195_dilations_0, groups = input_195_groups_0, pad = input_195_pad_0, pad_type = input_195_pad_type_0, strides = input_195_strides_0, weight = model_model_layers_10_mlp_gate_proj_weight_palettized, x = input_193)[name = string("input_195")]; + string b_21_pad_type_0 = const()[name = string("b_21_pad_type_0"), val = string("valid")]; + tensor b_21_strides_0 = const()[name = string("b_21_strides_0"), val = tensor([1, 1])]; + tensor b_21_pad_0 = const()[name = string("b_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_21_dilations_0 = const()[name = string("b_21_dilations_0"), val = tensor([1, 1])]; + int32 b_21_groups_0 = const()[name = string("b_21_groups_0"), val = int32(1)]; + tensor b_21 = conv(dilations = b_21_dilations_0, groups = b_21_groups_0, pad = b_21_pad_0, pad_type = b_21_pad_type_0, strides = b_21_strides_0, weight = model_model_layers_10_mlp_up_proj_weight_palettized, x = input_193)[name = string("b_21")]; + tensor c_21 = silu(x = input_195)[name = string("c_21")]; + tensor input_197 = mul(x = c_21, y = b_21)[name = string("input_197")]; + string e_21_pad_type_0 = const()[name = string("e_21_pad_type_0"), val = string("valid")]; + tensor e_21_strides_0 = const()[name = string("e_21_strides_0"), val = tensor([1, 1])]; + tensor e_21_pad_0 = const()[name = string("e_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_21_dilations_0 = const()[name = string("e_21_dilations_0"), val = tensor([1, 1])]; + int32 e_21_groups_0 = const()[name = string("e_21_groups_0"), val = int32(1)]; + tensor e_21 = conv(dilations = e_21_dilations_0, groups = e_21_groups_0, pad = e_21_pad_0, pad_type = e_21_pad_type_0, strides = e_21_strides_0, weight = model_model_layers_10_mlp_down_proj_weight_palettized, x = input_197)[name = string("e_21")]; + tensor var_6241_axes_0 = const()[name = string("op_6241_axes_0"), val = tensor([2])]; + tensor var_6241 = squeeze(axes = var_6241_axes_0, x = e_21)[name = string("op_6241")]; + tensor var_6242 = const()[name = string("op_6242"), val = tensor([0, 2, 1])]; + tensor var_6243 = transpose(perm = var_6242, x = var_6241)[name = string("transpose_18")]; + tensor hidden_states_111_cast_fp16 = add(x = hidden_states_109_cast_fp16, y = var_6243)[name = string("hidden_states_111_cast_fp16")]; + int32 var_6255 = const()[name = string("op_6255"), val = int32(-1)]; + fp16 const_330_promoted_to_fp16 = const()[name = string("const_330_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6257_cast_fp16 = mul(x = hidden_states_111_cast_fp16, y = const_330_promoted_to_fp16)[name = string("op_6257_cast_fp16")]; + bool input_199_interleave_0 = const()[name = string("input_199_interleave_0"), val = bool(false)]; + tensor input_199_cast_fp16 = concat(axis = var_6255, interleave = input_199_interleave_0, values = (hidden_states_111_cast_fp16, var_6257_cast_fp16))[name = string("input_199_cast_fp16")]; + tensor normed_177_axes_0 = const()[name = string("normed_177_axes_0"), val = tensor([-1])]; + fp16 var_6252_to_fp16 = const()[name = string("op_6252_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_177_cast_fp16 = layer_norm(axes = normed_177_axes_0, epsilon = var_6252_to_fp16, x = input_199_cast_fp16)[name = string("normed_177_cast_fp16")]; + tensor normed_179_begin_0 = const()[name = string("normed_179_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_179_end_0 = const()[name = string("normed_179_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_179_end_mask_0 = const()[name = string("normed_179_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_179_cast_fp16 = slice_by_index(begin = normed_179_begin_0, end = normed_179_end_0, end_mask = normed_179_end_mask_0, x = normed_177_cast_fp16)[name = string("normed_179_cast_fp16")]; + tensor const_333_promoted_to_fp16 = const()[name = string("const_333_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711176512)))]; + tensor hidden_states_113_cast_fp16 = mul(x = normed_179_cast_fp16, y = const_333_promoted_to_fp16)[name = string("hidden_states_113_cast_fp16")]; + tensor var_6274 = const()[name = string("op_6274"), val = tensor([0, 2, 1])]; + tensor var_6277_axes_0 = const()[name = string("op_6277_axes_0"), val = tensor([2])]; + tensor var_6275_cast_fp16 = transpose(perm = var_6274, x = hidden_states_113_cast_fp16)[name = string("transpose_17")]; + tensor var_6277_cast_fp16 = expand_dims(axes = var_6277_axes_0, x = var_6275_cast_fp16)[name = string("op_6277_cast_fp16")]; + string var_6293_pad_type_0 = const()[name = string("op_6293_pad_type_0"), val = string("valid")]; + tensor var_6293_strides_0 = const()[name = string("op_6293_strides_0"), val = tensor([1, 1])]; + tensor var_6293_pad_0 = const()[name = string("op_6293_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6293_dilations_0 = const()[name = string("op_6293_dilations_0"), val = tensor([1, 1])]; + int32 var_6293_groups_0 = const()[name = string("op_6293_groups_0"), val = int32(1)]; + tensor var_6293 = conv(dilations = var_6293_dilations_0, groups = var_6293_groups_0, pad = var_6293_pad_0, pad_type = var_6293_pad_type_0, strides = var_6293_strides_0, weight = model_model_layers_11_self_attn_q_proj_weight_palettized, x = var_6277_cast_fp16)[name = string("op_6293")]; + tensor var_6298 = const()[name = string("op_6298"), val = tensor([1, 16, 1, 128])]; + tensor var_6299 = reshape(shape = var_6298, x = var_6293)[name = string("op_6299")]; + string var_6315_pad_type_0 = const()[name = string("op_6315_pad_type_0"), val = string("valid")]; + tensor var_6315_strides_0 = const()[name = string("op_6315_strides_0"), val = tensor([1, 1])]; + tensor var_6315_pad_0 = const()[name = string("op_6315_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6315_dilations_0 = const()[name = string("op_6315_dilations_0"), val = tensor([1, 1])]; + int32 var_6315_groups_0 = const()[name = string("op_6315_groups_0"), val = int32(1)]; + tensor var_6315 = conv(dilations = var_6315_dilations_0, groups = var_6315_groups_0, pad = var_6315_pad_0, pad_type = var_6315_pad_type_0, strides = var_6315_strides_0, weight = model_model_layers_11_self_attn_k_proj_weight_palettized, x = var_6277_cast_fp16)[name = string("op_6315")]; + tensor var_6320 = const()[name = string("op_6320"), val = tensor([1, 8, 1, 128])]; + tensor var_6321 = reshape(shape = var_6320, x = var_6315)[name = string("op_6321")]; + string var_6337_pad_type_0 = const()[name = string("op_6337_pad_type_0"), val = string("valid")]; + tensor var_6337_strides_0 = const()[name = string("op_6337_strides_0"), val = tensor([1, 1])]; + tensor var_6337_pad_0 = const()[name = string("op_6337_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6337_dilations_0 = const()[name = string("op_6337_dilations_0"), val = tensor([1, 1])]; + int32 var_6337_groups_0 = const()[name = string("op_6337_groups_0"), val = int32(1)]; + tensor var_6337 = conv(dilations = var_6337_dilations_0, groups = var_6337_groups_0, pad = var_6337_pad_0, pad_type = var_6337_pad_type_0, strides = var_6337_strides_0, weight = model_model_layers_11_self_attn_v_proj_weight_palettized, x = var_6277_cast_fp16)[name = string("op_6337")]; + tensor var_6342 = const()[name = string("op_6342"), val = tensor([1, 8, 1, 128])]; + tensor var_6343 = reshape(shape = var_6342, x = var_6337)[name = string("op_6343")]; + int32 var_6358 = const()[name = string("op_6358"), val = int32(-1)]; + fp16 const_334_promoted = const()[name = string("const_334_promoted"), val = fp16(-0x1p+0)]; + tensor var_6360 = mul(x = var_6299, y = const_334_promoted)[name = string("op_6360")]; + bool input_203_interleave_0 = const()[name = string("input_203_interleave_0"), val = bool(false)]; + tensor input_203 = concat(axis = var_6358, interleave = input_203_interleave_0, values = (var_6299, var_6360))[name = string("input_203")]; + tensor normed_181_axes_0 = const()[name = string("normed_181_axes_0"), val = tensor([-1])]; + fp16 var_6355_to_fp16 = const()[name = string("op_6355_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_181_cast_fp16 = layer_norm(axes = normed_181_axes_0, epsilon = var_6355_to_fp16, x = input_203)[name = string("normed_181_cast_fp16")]; + tensor normed_183_begin_0 = const()[name = string("normed_183_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_183_end_0 = const()[name = string("normed_183_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_183_end_mask_0 = const()[name = string("normed_183_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_183 = slice_by_index(begin = normed_183_begin_0, end = normed_183_end_0, end_mask = normed_183_end_mask_0, x = normed_181_cast_fp16)[name = string("normed_183")]; + tensor const_337 = const()[name = string("const_337"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711180672)))]; + tensor q_23 = mul(x = normed_183, y = const_337)[name = string("q_23")]; + int32 var_6383 = const()[name = string("op_6383"), val = int32(-1)]; + fp16 const_338_promoted = const()[name = string("const_338_promoted"), val = fp16(-0x1p+0)]; + tensor var_6385 = mul(x = var_6321, y = const_338_promoted)[name = string("op_6385")]; + bool input_205_interleave_0 = const()[name = string("input_205_interleave_0"), val = bool(false)]; + tensor input_205 = concat(axis = var_6383, interleave = input_205_interleave_0, values = (var_6321, var_6385))[name = string("input_205")]; + tensor normed_185_axes_0 = const()[name = string("normed_185_axes_0"), val = tensor([-1])]; + fp16 var_6380_to_fp16 = const()[name = string("op_6380_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_185_cast_fp16 = layer_norm(axes = normed_185_axes_0, epsilon = var_6380_to_fp16, x = input_205)[name = string("normed_185_cast_fp16")]; + tensor normed_187_begin_0 = const()[name = string("normed_187_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_187_end_0 = const()[name = string("normed_187_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_187_end_mask_0 = const()[name = string("normed_187_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_187 = slice_by_index(begin = normed_187_begin_0, end = normed_187_end_0, end_mask = normed_187_end_mask_0, x = normed_185_cast_fp16)[name = string("normed_187")]; + tensor const_341 = const()[name = string("const_341"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711180992)))]; + tensor k_23 = mul(x = normed_187, y = const_341)[name = string("k_23")]; + tensor var_6399 = mul(x = q_23, y = cos_1_cast_fp16)[name = string("op_6399")]; + tensor x1_45_begin_0 = const()[name = string("x1_45_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_45_end_0 = const()[name = string("x1_45_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_45_end_mask_0 = const()[name = string("x1_45_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_45 = slice_by_index(begin = x1_45_begin_0, end = x1_45_end_0, end_mask = x1_45_end_mask_0, x = q_23)[name = string("x1_45")]; + tensor x2_45_begin_0 = const()[name = string("x2_45_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_45_end_0 = const()[name = string("x2_45_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_45_end_mask_0 = const()[name = string("x2_45_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_45 = slice_by_index(begin = x2_45_begin_0, end = x2_45_end_0, end_mask = x2_45_end_mask_0, x = q_23)[name = string("x2_45")]; + fp16 const_344_promoted = const()[name = string("const_344_promoted"), val = fp16(-0x1p+0)]; + tensor var_6420 = mul(x = x2_45, y = const_344_promoted)[name = string("op_6420")]; + int32 var_6422 = const()[name = string("op_6422"), val = int32(-1)]; + bool var_6423_interleave_0 = const()[name = string("op_6423_interleave_0"), val = bool(false)]; + tensor var_6423 = concat(axis = var_6422, interleave = var_6423_interleave_0, values = (var_6420, x1_45))[name = string("op_6423")]; + tensor var_6424 = mul(x = var_6423, y = sin_1_cast_fp16)[name = string("op_6424")]; + tensor query_states_45 = add(x = var_6399, y = var_6424)[name = string("query_states_45")]; + tensor var_6427 = mul(x = k_23, y = cos_1_cast_fp16)[name = string("op_6427")]; + tensor x1_47_begin_0 = const()[name = string("x1_47_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_47_end_0 = const()[name = string("x1_47_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_47_end_mask_0 = const()[name = string("x1_47_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_47 = slice_by_index(begin = x1_47_begin_0, end = x1_47_end_0, end_mask = x1_47_end_mask_0, x = k_23)[name = string("x1_47")]; + tensor x2_47_begin_0 = const()[name = string("x2_47_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_47_end_0 = const()[name = string("x2_47_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_47_end_mask_0 = const()[name = string("x2_47_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_47 = slice_by_index(begin = x2_47_begin_0, end = x2_47_end_0, end_mask = x2_47_end_mask_0, x = k_23)[name = string("x2_47")]; + fp16 const_347_promoted = const()[name = string("const_347_promoted"), val = fp16(-0x1p+0)]; + tensor var_6448 = mul(x = x2_47, y = const_347_promoted)[name = string("op_6448")]; + int32 var_6450 = const()[name = string("op_6450"), val = int32(-1)]; + bool var_6451_interleave_0 = const()[name = string("op_6451_interleave_0"), val = bool(false)]; + tensor var_6451 = concat(axis = var_6450, interleave = var_6451_interleave_0, values = (var_6448, x1_47))[name = string("op_6451")]; + tensor var_6452 = mul(x = var_6451, y = sin_1_cast_fp16)[name = string("op_6452")]; + tensor key_states_45 = add(x = var_6427, y = var_6452)[name = string("key_states_45")]; + tensor expand_dims_132 = const()[name = string("expand_dims_132"), val = tensor([11])]; + tensor expand_dims_133 = const()[name = string("expand_dims_133"), val = tensor([0])]; + tensor expand_dims_135 = const()[name = string("expand_dims_135"), val = tensor([0])]; + tensor expand_dims_136 = const()[name = string("expand_dims_136"), val = tensor([12])]; + int32 concat_90_axis_0 = const()[name = string("concat_90_axis_0"), val = int32(0)]; + bool concat_90_interleave_0 = const()[name = string("concat_90_interleave_0"), val = bool(false)]; + tensor concat_90 = concat(axis = concat_90_axis_0, interleave = concat_90_interleave_0, values = (expand_dims_132, expand_dims_133, current_pos, expand_dims_135))[name = string("concat_90")]; + tensor concat_91_values1_0 = const()[name = string("concat_91_values1_0"), val = tensor([0])]; + tensor concat_91_values3_0 = const()[name = string("concat_91_values3_0"), val = tensor([0])]; + int32 concat_91_axis_0 = const()[name = string("concat_91_axis_0"), val = int32(0)]; + bool concat_91_interleave_0 = const()[name = string("concat_91_interleave_0"), val = bool(false)]; + tensor concat_91 = concat(axis = concat_91_axis_0, interleave = concat_91_interleave_0, values = (expand_dims_136, concat_91_values1_0, var_1001, concat_91_values3_0))[name = string("concat_91")]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_90, begin_mask = model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0, end = concat_91, end_mask = model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_23_stride_0, update = key_states_45, x = coreml_update_state_49)[name = string("model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_78_write_state")]; + tensor coreml_update_state_50 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_78")]; + tensor expand_dims_138 = const()[name = string("expand_dims_138"), val = tensor([39])]; + tensor expand_dims_139 = const()[name = string("expand_dims_139"), val = tensor([0])]; + tensor expand_dims_141 = const()[name = string("expand_dims_141"), val = tensor([0])]; + tensor expand_dims_142 = const()[name = string("expand_dims_142"), val = tensor([40])]; + int32 concat_94_axis_0 = const()[name = string("concat_94_axis_0"), val = int32(0)]; + bool concat_94_interleave_0 = const()[name = string("concat_94_interleave_0"), val = bool(false)]; + tensor concat_94 = concat(axis = concat_94_axis_0, interleave = concat_94_interleave_0, values = (expand_dims_138, expand_dims_139, current_pos, expand_dims_141))[name = string("concat_94")]; + tensor concat_95_values1_0 = const()[name = string("concat_95_values1_0"), val = tensor([0])]; + tensor concat_95_values3_0 = const()[name = string("concat_95_values3_0"), val = tensor([0])]; + int32 concat_95_axis_0 = const()[name = string("concat_95_axis_0"), val = int32(0)]; + bool concat_95_interleave_0 = const()[name = string("concat_95_interleave_0"), val = bool(false)]; + tensor concat_95 = concat(axis = concat_95_axis_0, interleave = concat_95_interleave_0, values = (expand_dims_142, concat_95_values1_0, var_1001, concat_95_values3_0))[name = string("concat_95")]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_94, begin_mask = model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0, end = concat_95, end_mask = model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_24_stride_0, update = var_6343, x = coreml_update_state_50)[name = string("model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_79_write_state")]; + tensor coreml_update_state_51 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_79")]; + tensor var_6507_begin_0 = const()[name = string("op_6507_begin_0"), val = tensor([11, 0, 0, 0])]; + tensor var_6507_end_0 = const()[name = string("op_6507_end_0"), val = tensor([12, 8, 1024, 128])]; + tensor var_6507_end_mask_0 = const()[name = string("op_6507_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_6507_cast_fp16 = slice_by_index(begin = var_6507_begin_0, end = var_6507_end_0, end_mask = var_6507_end_mask_0, x = coreml_update_state_51)[name = string("op_6507_cast_fp16")]; + tensor K_layer_cache_23_axes_0 = const()[name = string("K_layer_cache_23_axes_0"), val = tensor([0])]; + tensor K_layer_cache_23_cast_fp16 = squeeze(axes = K_layer_cache_23_axes_0, x = var_6507_cast_fp16)[name = string("K_layer_cache_23_cast_fp16")]; + tensor var_6514_begin_0 = const()[name = string("op_6514_begin_0"), val = tensor([39, 0, 0, 0])]; + tensor var_6514_end_0 = const()[name = string("op_6514_end_0"), val = tensor([40, 8, 1024, 128])]; + tensor var_6514_end_mask_0 = const()[name = string("op_6514_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_6514_cast_fp16 = slice_by_index(begin = var_6514_begin_0, end = var_6514_end_0, end_mask = var_6514_end_mask_0, x = coreml_update_state_51)[name = string("op_6514_cast_fp16")]; + tensor V_layer_cache_23_axes_0 = const()[name = string("V_layer_cache_23_axes_0"), val = tensor([0])]; + tensor V_layer_cache_23_cast_fp16 = squeeze(axes = V_layer_cache_23_axes_0, x = var_6514_cast_fp16)[name = string("V_layer_cache_23_cast_fp16")]; + tensor x_179_axes_0 = const()[name = string("x_179_axes_0"), val = tensor([1])]; + tensor x_179_cast_fp16 = expand_dims(axes = x_179_axes_0, x = K_layer_cache_23_cast_fp16)[name = string("x_179_cast_fp16")]; + tensor var_6551 = const()[name = string("op_6551"), val = tensor([1, 2, 1, 1])]; + tensor x_181_cast_fp16 = tile(reps = var_6551, x = x_179_cast_fp16)[name = string("x_181_cast_fp16")]; + tensor var_6563 = const()[name = string("op_6563"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_47_cast_fp16 = reshape(shape = var_6563, x = x_181_cast_fp16)[name = string("key_states_47_cast_fp16")]; + tensor x_185_axes_0 = const()[name = string("x_185_axes_0"), val = tensor([1])]; + tensor x_185_cast_fp16 = expand_dims(axes = x_185_axes_0, x = V_layer_cache_23_cast_fp16)[name = string("x_185_cast_fp16")]; + tensor var_6571 = const()[name = string("op_6571"), val = tensor([1, 2, 1, 1])]; + tensor x_187_cast_fp16 = tile(reps = var_6571, x = x_185_cast_fp16)[name = string("x_187_cast_fp16")]; + tensor var_6583 = const()[name = string("op_6583"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_69_cast_fp16 = reshape(shape = var_6583, x = x_187_cast_fp16)[name = string("value_states_69_cast_fp16")]; + bool var_6598_transpose_x_1 = const()[name = string("op_6598_transpose_x_1"), val = bool(false)]; + bool var_6598_transpose_y_1 = const()[name = string("op_6598_transpose_y_1"), val = bool(true)]; + tensor var_6598 = matmul(transpose_x = var_6598_transpose_x_1, transpose_y = var_6598_transpose_y_1, x = query_states_45, y = key_states_47_cast_fp16)[name = string("op_6598")]; + fp16 var_6599_to_fp16 = const()[name = string("op_6599_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_67_cast_fp16 = mul(x = var_6598, y = var_6599_to_fp16)[name = string("attn_weights_67_cast_fp16")]; + tensor attn_weights_69_cast_fp16 = add(x = attn_weights_67_cast_fp16, y = causal_mask)[name = string("attn_weights_69_cast_fp16")]; + int32 var_6634 = const()[name = string("op_6634"), val = int32(-1)]; + tensor attn_weights_71_cast_fp16 = softmax(axis = var_6634, x = attn_weights_69_cast_fp16)[name = string("attn_weights_71_cast_fp16")]; + bool attn_output_111_transpose_x_0 = const()[name = string("attn_output_111_transpose_x_0"), val = bool(false)]; + bool attn_output_111_transpose_y_0 = const()[name = string("attn_output_111_transpose_y_0"), val = bool(false)]; + tensor attn_output_111_cast_fp16 = matmul(transpose_x = attn_output_111_transpose_x_0, transpose_y = attn_output_111_transpose_y_0, x = attn_weights_71_cast_fp16, y = value_states_69_cast_fp16)[name = string("attn_output_111_cast_fp16")]; + tensor var_6645_perm_0 = const()[name = string("op_6645_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_6649 = const()[name = string("op_6649"), val = tensor([1, 1, 2048])]; + tensor var_6645_cast_fp16 = transpose(perm = var_6645_perm_0, x = attn_output_111_cast_fp16)[name = string("transpose_16")]; + tensor attn_output_115_cast_fp16 = reshape(shape = var_6649, x = var_6645_cast_fp16)[name = string("attn_output_115_cast_fp16")]; + tensor var_6654 = const()[name = string("op_6654"), val = tensor([0, 2, 1])]; + string var_6670_pad_type_0 = const()[name = string("op_6670_pad_type_0"), val = string("valid")]; + int32 var_6670_groups_0 = const()[name = string("op_6670_groups_0"), val = int32(1)]; + tensor var_6670_strides_0 = const()[name = string("op_6670_strides_0"), val = tensor([1])]; + tensor var_6670_pad_0 = const()[name = string("op_6670_pad_0"), val = tensor([0, 0])]; + tensor var_6670_dilations_0 = const()[name = string("op_6670_dilations_0"), val = tensor([1])]; + tensor squeeze_11_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711181312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715375680))))[name = string("squeeze_11_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_6655_cast_fp16 = transpose(perm = var_6654, x = attn_output_115_cast_fp16)[name = string("transpose_15")]; + tensor var_6670_cast_fp16 = conv(dilations = var_6670_dilations_0, groups = var_6670_groups_0, pad = var_6670_pad_0, pad_type = var_6670_pad_type_0, strides = var_6670_strides_0, weight = squeeze_11_cast_fp16_to_fp32_to_fp16_palettized, x = var_6655_cast_fp16)[name = string("op_6670_cast_fp16")]; + tensor var_6674 = const()[name = string("op_6674"), val = tensor([0, 2, 1])]; + tensor attn_output_119_cast_fp16 = transpose(perm = var_6674, x = var_6670_cast_fp16)[name = string("transpose_14")]; + tensor hidden_states_119_cast_fp16 = add(x = hidden_states_111_cast_fp16, y = attn_output_119_cast_fp16)[name = string("hidden_states_119_cast_fp16")]; + int32 var_6687 = const()[name = string("op_6687"), val = int32(-1)]; + fp16 const_356_promoted_to_fp16 = const()[name = string("const_356_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6689_cast_fp16 = mul(x = hidden_states_119_cast_fp16, y = const_356_promoted_to_fp16)[name = string("op_6689_cast_fp16")]; + bool input_209_interleave_0 = const()[name = string("input_209_interleave_0"), val = bool(false)]; + tensor input_209_cast_fp16 = concat(axis = var_6687, interleave = input_209_interleave_0, values = (hidden_states_119_cast_fp16, var_6689_cast_fp16))[name = string("input_209_cast_fp16")]; + tensor normed_189_axes_0 = const()[name = string("normed_189_axes_0"), val = tensor([-1])]; + fp16 var_6684_to_fp16 = const()[name = string("op_6684_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_189_cast_fp16 = layer_norm(axes = normed_189_axes_0, epsilon = var_6684_to_fp16, x = input_209_cast_fp16)[name = string("normed_189_cast_fp16")]; + tensor normed_191_begin_0 = const()[name = string("normed_191_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_191_end_0 = const()[name = string("normed_191_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_191_end_mask_0 = const()[name = string("normed_191_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_191_cast_fp16 = slice_by_index(begin = normed_191_begin_0, end = normed_191_end_0, end_mask = normed_191_end_mask_0, x = normed_189_cast_fp16)[name = string("normed_191_cast_fp16")]; + tensor const_359_promoted_to_fp16 = const()[name = string("const_359_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715506816)))]; + tensor x_189_cast_fp16 = mul(x = normed_191_cast_fp16, y = const_359_promoted_to_fp16)[name = string("x_189_cast_fp16")]; + tensor var_6714 = const()[name = string("op_6714"), val = tensor([0, 2, 1])]; + tensor input_211_axes_0 = const()[name = string("input_211_axes_0"), val = tensor([2])]; + tensor var_6715 = transpose(perm = var_6714, x = x_189_cast_fp16)[name = string("transpose_13")]; + tensor input_211 = expand_dims(axes = input_211_axes_0, x = var_6715)[name = string("input_211")]; + string input_213_pad_type_0 = const()[name = string("input_213_pad_type_0"), val = string("valid")]; + tensor input_213_strides_0 = const()[name = string("input_213_strides_0"), val = tensor([1, 1])]; + tensor input_213_pad_0 = const()[name = string("input_213_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_213_dilations_0 = const()[name = string("input_213_dilations_0"), val = tensor([1, 1])]; + int32 input_213_groups_0 = const()[name = string("input_213_groups_0"), val = int32(1)]; + tensor input_213 = conv(dilations = input_213_dilations_0, groups = input_213_groups_0, pad = input_213_pad_0, pad_type = input_213_pad_type_0, strides = input_213_strides_0, weight = model_model_layers_11_mlp_gate_proj_weight_palettized, x = input_211)[name = string("input_213")]; + string b_23_pad_type_0 = const()[name = string("b_23_pad_type_0"), val = string("valid")]; + tensor b_23_strides_0 = const()[name = string("b_23_strides_0"), val = tensor([1, 1])]; + tensor b_23_pad_0 = const()[name = string("b_23_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_23_dilations_0 = const()[name = string("b_23_dilations_0"), val = tensor([1, 1])]; + int32 b_23_groups_0 = const()[name = string("b_23_groups_0"), val = int32(1)]; + tensor b_23 = conv(dilations = b_23_dilations_0, groups = b_23_groups_0, pad = b_23_pad_0, pad_type = b_23_pad_type_0, strides = b_23_strides_0, weight = model_model_layers_11_mlp_up_proj_weight_palettized, x = input_211)[name = string("b_23")]; + tensor c_23 = silu(x = input_213)[name = string("c_23")]; + tensor input_215 = mul(x = c_23, y = b_23)[name = string("input_215")]; + string e_23_pad_type_0 = const()[name = string("e_23_pad_type_0"), val = string("valid")]; + tensor e_23_strides_0 = const()[name = string("e_23_strides_0"), val = tensor([1, 1])]; + tensor e_23_pad_0 = const()[name = string("e_23_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_23_dilations_0 = const()[name = string("e_23_dilations_0"), val = tensor([1, 1])]; + int32 e_23_groups_0 = const()[name = string("e_23_groups_0"), val = int32(1)]; + tensor e_23 = conv(dilations = e_23_dilations_0, groups = e_23_groups_0, pad = e_23_pad_0, pad_type = e_23_pad_type_0, strides = e_23_strides_0, weight = model_model_layers_11_mlp_down_proj_weight_palettized, x = input_215)[name = string("e_23")]; + tensor var_6737_axes_0 = const()[name = string("op_6737_axes_0"), val = tensor([2])]; + tensor var_6737 = squeeze(axes = var_6737_axes_0, x = e_23)[name = string("op_6737")]; + tensor var_6738 = const()[name = string("op_6738"), val = tensor([0, 2, 1])]; + tensor var_6739 = transpose(perm = var_6738, x = var_6737)[name = string("transpose_12")]; + tensor hidden_states_121_cast_fp16 = add(x = hidden_states_119_cast_fp16, y = var_6739)[name = string("hidden_states_121_cast_fp16")]; + int32 var_6751 = const()[name = string("op_6751"), val = int32(-1)]; + fp16 const_360_promoted_to_fp16 = const()[name = string("const_360_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6753_cast_fp16 = mul(x = hidden_states_121_cast_fp16, y = const_360_promoted_to_fp16)[name = string("op_6753_cast_fp16")]; + bool input_217_interleave_0 = const()[name = string("input_217_interleave_0"), val = bool(false)]; + tensor input_217_cast_fp16 = concat(axis = var_6751, interleave = input_217_interleave_0, values = (hidden_states_121_cast_fp16, var_6753_cast_fp16))[name = string("input_217_cast_fp16")]; + tensor normed_193_axes_0 = const()[name = string("normed_193_axes_0"), val = tensor([-1])]; + fp16 var_6748_to_fp16 = const()[name = string("op_6748_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_193_cast_fp16 = layer_norm(axes = normed_193_axes_0, epsilon = var_6748_to_fp16, x = input_217_cast_fp16)[name = string("normed_193_cast_fp16")]; + tensor normed_195_begin_0 = const()[name = string("normed_195_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_195_end_0 = const()[name = string("normed_195_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_195_end_mask_0 = const()[name = string("normed_195_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_195_cast_fp16 = slice_by_index(begin = normed_195_begin_0, end = normed_195_end_0, end_mask = normed_195_end_mask_0, x = normed_193_cast_fp16)[name = string("normed_195_cast_fp16")]; + tensor const_363_promoted_to_fp16 = const()[name = string("const_363_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715510976)))]; + tensor hidden_states_123_cast_fp16 = mul(x = normed_195_cast_fp16, y = const_363_promoted_to_fp16)[name = string("hidden_states_123_cast_fp16")]; + tensor var_6770 = const()[name = string("op_6770"), val = tensor([0, 2, 1])]; + tensor var_6773_axes_0 = const()[name = string("op_6773_axes_0"), val = tensor([2])]; + tensor var_6771_cast_fp16 = transpose(perm = var_6770, x = hidden_states_123_cast_fp16)[name = string("transpose_11")]; + tensor var_6773_cast_fp16 = expand_dims(axes = var_6773_axes_0, x = var_6771_cast_fp16)[name = string("op_6773_cast_fp16")]; + string var_6789_pad_type_0 = const()[name = string("op_6789_pad_type_0"), val = string("valid")]; + tensor var_6789_strides_0 = const()[name = string("op_6789_strides_0"), val = tensor([1, 1])]; + tensor var_6789_pad_0 = const()[name = string("op_6789_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6789_dilations_0 = const()[name = string("op_6789_dilations_0"), val = tensor([1, 1])]; + int32 var_6789_groups_0 = const()[name = string("op_6789_groups_0"), val = int32(1)]; + tensor var_6789 = conv(dilations = var_6789_dilations_0, groups = var_6789_groups_0, pad = var_6789_pad_0, pad_type = var_6789_pad_type_0, strides = var_6789_strides_0, weight = model_model_layers_12_self_attn_q_proj_weight_palettized, x = var_6773_cast_fp16)[name = string("op_6789")]; + tensor var_6794 = const()[name = string("op_6794"), val = tensor([1, 16, 1, 128])]; + tensor var_6795 = reshape(shape = var_6794, x = var_6789)[name = string("op_6795")]; + string var_6811_pad_type_0 = const()[name = string("op_6811_pad_type_0"), val = string("valid")]; + tensor var_6811_strides_0 = const()[name = string("op_6811_strides_0"), val = tensor([1, 1])]; + tensor var_6811_pad_0 = const()[name = string("op_6811_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6811_dilations_0 = const()[name = string("op_6811_dilations_0"), val = tensor([1, 1])]; + int32 var_6811_groups_0 = const()[name = string("op_6811_groups_0"), val = int32(1)]; + tensor var_6811 = conv(dilations = var_6811_dilations_0, groups = var_6811_groups_0, pad = var_6811_pad_0, pad_type = var_6811_pad_type_0, strides = var_6811_strides_0, weight = model_model_layers_12_self_attn_k_proj_weight_palettized, x = var_6773_cast_fp16)[name = string("op_6811")]; + tensor var_6816 = const()[name = string("op_6816"), val = tensor([1, 8, 1, 128])]; + tensor var_6817 = reshape(shape = var_6816, x = var_6811)[name = string("op_6817")]; + string var_6833_pad_type_0 = const()[name = string("op_6833_pad_type_0"), val = string("valid")]; + tensor var_6833_strides_0 = const()[name = string("op_6833_strides_0"), val = tensor([1, 1])]; + tensor var_6833_pad_0 = const()[name = string("op_6833_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6833_dilations_0 = const()[name = string("op_6833_dilations_0"), val = tensor([1, 1])]; + int32 var_6833_groups_0 = const()[name = string("op_6833_groups_0"), val = int32(1)]; + tensor var_6833 = conv(dilations = var_6833_dilations_0, groups = var_6833_groups_0, pad = var_6833_pad_0, pad_type = var_6833_pad_type_0, strides = var_6833_strides_0, weight = model_model_layers_12_self_attn_v_proj_weight_palettized, x = var_6773_cast_fp16)[name = string("op_6833")]; + tensor var_6838 = const()[name = string("op_6838"), val = tensor([1, 8, 1, 128])]; + tensor var_6839 = reshape(shape = var_6838, x = var_6833)[name = string("op_6839")]; + int32 var_6854 = const()[name = string("op_6854"), val = int32(-1)]; + fp16 const_364_promoted = const()[name = string("const_364_promoted"), val = fp16(-0x1p+0)]; + tensor var_6856 = mul(x = var_6795, y = const_364_promoted)[name = string("op_6856")]; + bool input_221_interleave_0 = const()[name = string("input_221_interleave_0"), val = bool(false)]; + tensor input_221 = concat(axis = var_6854, interleave = input_221_interleave_0, values = (var_6795, var_6856))[name = string("input_221")]; + tensor normed_197_axes_0 = const()[name = string("normed_197_axes_0"), val = tensor([-1])]; + fp16 var_6851_to_fp16 = const()[name = string("op_6851_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_197_cast_fp16 = layer_norm(axes = normed_197_axes_0, epsilon = var_6851_to_fp16, x = input_221)[name = string("normed_197_cast_fp16")]; + tensor normed_199_begin_0 = const()[name = string("normed_199_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_199_end_0 = const()[name = string("normed_199_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_199_end_mask_0 = const()[name = string("normed_199_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_199 = slice_by_index(begin = normed_199_begin_0, end = normed_199_end_0, end_mask = normed_199_end_mask_0, x = normed_197_cast_fp16)[name = string("normed_199")]; + tensor const_367 = const()[name = string("const_367"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715515136)))]; + tensor q_25 = mul(x = normed_199, y = const_367)[name = string("q_25")]; + int32 var_6879 = const()[name = string("op_6879"), val = int32(-1)]; + fp16 const_368_promoted = const()[name = string("const_368_promoted"), val = fp16(-0x1p+0)]; + tensor var_6881 = mul(x = var_6817, y = const_368_promoted)[name = string("op_6881")]; + bool input_223_interleave_0 = const()[name = string("input_223_interleave_0"), val = bool(false)]; + tensor input_223 = concat(axis = var_6879, interleave = input_223_interleave_0, values = (var_6817, var_6881))[name = string("input_223")]; + tensor normed_201_axes_0 = const()[name = string("normed_201_axes_0"), val = tensor([-1])]; + fp16 var_6876_to_fp16 = const()[name = string("op_6876_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_201_cast_fp16 = layer_norm(axes = normed_201_axes_0, epsilon = var_6876_to_fp16, x = input_223)[name = string("normed_201_cast_fp16")]; + tensor normed_203_begin_0 = const()[name = string("normed_203_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_203_end_0 = const()[name = string("normed_203_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_203_end_mask_0 = const()[name = string("normed_203_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_203 = slice_by_index(begin = normed_203_begin_0, end = normed_203_end_0, end_mask = normed_203_end_mask_0, x = normed_201_cast_fp16)[name = string("normed_203")]; + tensor const_371 = const()[name = string("const_371"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715515456)))]; + tensor k_25 = mul(x = normed_203, y = const_371)[name = string("k_25")]; + tensor var_6895 = mul(x = q_25, y = cos_1_cast_fp16)[name = string("op_6895")]; + tensor x1_49_begin_0 = const()[name = string("x1_49_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_49_end_0 = const()[name = string("x1_49_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_49_end_mask_0 = const()[name = string("x1_49_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_49 = slice_by_index(begin = x1_49_begin_0, end = x1_49_end_0, end_mask = x1_49_end_mask_0, x = q_25)[name = string("x1_49")]; + tensor x2_49_begin_0 = const()[name = string("x2_49_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_49_end_0 = const()[name = string("x2_49_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_49_end_mask_0 = const()[name = string("x2_49_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_49 = slice_by_index(begin = x2_49_begin_0, end = x2_49_end_0, end_mask = x2_49_end_mask_0, x = q_25)[name = string("x2_49")]; + fp16 const_374_promoted = const()[name = string("const_374_promoted"), val = fp16(-0x1p+0)]; + tensor var_6916 = mul(x = x2_49, y = const_374_promoted)[name = string("op_6916")]; + int32 var_6918 = const()[name = string("op_6918"), val = int32(-1)]; + bool var_6919_interleave_0 = const()[name = string("op_6919_interleave_0"), val = bool(false)]; + tensor var_6919 = concat(axis = var_6918, interleave = var_6919_interleave_0, values = (var_6916, x1_49))[name = string("op_6919")]; + tensor var_6920 = mul(x = var_6919, y = sin_1_cast_fp16)[name = string("op_6920")]; + tensor query_states_49 = add(x = var_6895, y = var_6920)[name = string("query_states_49")]; + tensor var_6923 = mul(x = k_25, y = cos_1_cast_fp16)[name = string("op_6923")]; + tensor x1_51_begin_0 = const()[name = string("x1_51_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_51_end_0 = const()[name = string("x1_51_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_51_end_mask_0 = const()[name = string("x1_51_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_51 = slice_by_index(begin = x1_51_begin_0, end = x1_51_end_0, end_mask = x1_51_end_mask_0, x = k_25)[name = string("x1_51")]; + tensor x2_51_begin_0 = const()[name = string("x2_51_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_51_end_0 = const()[name = string("x2_51_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_51_end_mask_0 = const()[name = string("x2_51_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_51 = slice_by_index(begin = x2_51_begin_0, end = x2_51_end_0, end_mask = x2_51_end_mask_0, x = k_25)[name = string("x2_51")]; + fp16 const_377_promoted = const()[name = string("const_377_promoted"), val = fp16(-0x1p+0)]; + tensor var_6944 = mul(x = x2_51, y = const_377_promoted)[name = string("op_6944")]; + int32 var_6946 = const()[name = string("op_6946"), val = int32(-1)]; + bool var_6947_interleave_0 = const()[name = string("op_6947_interleave_0"), val = bool(false)]; + tensor var_6947 = concat(axis = var_6946, interleave = var_6947_interleave_0, values = (var_6944, x1_51))[name = string("op_6947")]; + tensor var_6948 = mul(x = var_6947, y = sin_1_cast_fp16)[name = string("op_6948")]; + tensor key_states_49 = add(x = var_6923, y = var_6948)[name = string("key_states_49")]; + tensor expand_dims_144 = const()[name = string("expand_dims_144"), val = tensor([12])]; + tensor expand_dims_145 = const()[name = string("expand_dims_145"), val = tensor([0])]; + tensor expand_dims_147 = const()[name = string("expand_dims_147"), val = tensor([0])]; + tensor expand_dims_148 = const()[name = string("expand_dims_148"), val = tensor([13])]; + int32 concat_98_axis_0 = const()[name = string("concat_98_axis_0"), val = int32(0)]; + bool concat_98_interleave_0 = const()[name = string("concat_98_interleave_0"), val = bool(false)]; + tensor concat_98 = concat(axis = concat_98_axis_0, interleave = concat_98_interleave_0, values = (expand_dims_144, expand_dims_145, current_pos, expand_dims_147))[name = string("concat_98")]; + tensor concat_99_values1_0 = const()[name = string("concat_99_values1_0"), val = tensor([0])]; + tensor concat_99_values3_0 = const()[name = string("concat_99_values3_0"), val = tensor([0])]; + int32 concat_99_axis_0 = const()[name = string("concat_99_axis_0"), val = int32(0)]; + bool concat_99_interleave_0 = const()[name = string("concat_99_interleave_0"), val = bool(false)]; + tensor concat_99 = concat(axis = concat_99_axis_0, interleave = concat_99_interleave_0, values = (expand_dims_148, concat_99_values1_0, var_1001, concat_99_values3_0))[name = string("concat_99")]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_98, begin_mask = model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0, end = concat_99, end_mask = model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_25_stride_0, update = key_states_49, x = coreml_update_state_51)[name = string("model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_80_write_state")]; + tensor coreml_update_state_52 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_80")]; + tensor expand_dims_150 = const()[name = string("expand_dims_150"), val = tensor([40])]; + tensor expand_dims_151 = const()[name = string("expand_dims_151"), val = tensor([0])]; + tensor expand_dims_153 = const()[name = string("expand_dims_153"), val = tensor([0])]; + tensor expand_dims_154 = const()[name = string("expand_dims_154"), val = tensor([41])]; + int32 concat_102_axis_0 = const()[name = string("concat_102_axis_0"), val = int32(0)]; + bool concat_102_interleave_0 = const()[name = string("concat_102_interleave_0"), val = bool(false)]; + tensor concat_102 = concat(axis = concat_102_axis_0, interleave = concat_102_interleave_0, values = (expand_dims_150, expand_dims_151, current_pos, expand_dims_153))[name = string("concat_102")]; + tensor concat_103_values1_0 = const()[name = string("concat_103_values1_0"), val = tensor([0])]; + tensor concat_103_values3_0 = const()[name = string("concat_103_values3_0"), val = tensor([0])]; + int32 concat_103_axis_0 = const()[name = string("concat_103_axis_0"), val = int32(0)]; + bool concat_103_interleave_0 = const()[name = string("concat_103_interleave_0"), val = bool(false)]; + tensor concat_103 = concat(axis = concat_103_axis_0, interleave = concat_103_interleave_0, values = (expand_dims_154, concat_103_values1_0, var_1001, concat_103_values3_0))[name = string("concat_103")]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_102, begin_mask = model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0, end = concat_103, end_mask = model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_26_stride_0, update = var_6839, x = coreml_update_state_52)[name = string("model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_81_write_state")]; + tensor coreml_update_state_53 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_81")]; + tensor var_7003_begin_0 = const()[name = string("op_7003_begin_0"), val = tensor([12, 0, 0, 0])]; + tensor var_7003_end_0 = const()[name = string("op_7003_end_0"), val = tensor([13, 8, 1024, 128])]; + tensor var_7003_end_mask_0 = const()[name = string("op_7003_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_7003_cast_fp16 = slice_by_index(begin = var_7003_begin_0, end = var_7003_end_0, end_mask = var_7003_end_mask_0, x = coreml_update_state_53)[name = string("op_7003_cast_fp16")]; + tensor K_layer_cache_25_axes_0 = const()[name = string("K_layer_cache_25_axes_0"), val = tensor([0])]; + tensor K_layer_cache_25_cast_fp16 = squeeze(axes = K_layer_cache_25_axes_0, x = var_7003_cast_fp16)[name = string("K_layer_cache_25_cast_fp16")]; + tensor var_7010_begin_0 = const()[name = string("op_7010_begin_0"), val = tensor([40, 0, 0, 0])]; + tensor var_7010_end_0 = const()[name = string("op_7010_end_0"), val = tensor([41, 8, 1024, 128])]; + tensor var_7010_end_mask_0 = const()[name = string("op_7010_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_7010_cast_fp16 = slice_by_index(begin = var_7010_begin_0, end = var_7010_end_0, end_mask = var_7010_end_mask_0, x = coreml_update_state_53)[name = string("op_7010_cast_fp16")]; + tensor V_layer_cache_25_axes_0 = const()[name = string("V_layer_cache_25_axes_0"), val = tensor([0])]; + tensor V_layer_cache_25_cast_fp16 = squeeze(axes = V_layer_cache_25_axes_0, x = var_7010_cast_fp16)[name = string("V_layer_cache_25_cast_fp16")]; + tensor x_195_axes_0 = const()[name = string("x_195_axes_0"), val = tensor([1])]; + tensor x_195_cast_fp16 = expand_dims(axes = x_195_axes_0, x = K_layer_cache_25_cast_fp16)[name = string("x_195_cast_fp16")]; + tensor var_7047 = const()[name = string("op_7047"), val = tensor([1, 2, 1, 1])]; + tensor x_197_cast_fp16 = tile(reps = var_7047, x = x_195_cast_fp16)[name = string("x_197_cast_fp16")]; + tensor var_7059 = const()[name = string("op_7059"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_51_cast_fp16 = reshape(shape = var_7059, x = x_197_cast_fp16)[name = string("key_states_51_cast_fp16")]; + tensor x_201_axes_0 = const()[name = string("x_201_axes_0"), val = tensor([1])]; + tensor x_201_cast_fp16 = expand_dims(axes = x_201_axes_0, x = V_layer_cache_25_cast_fp16)[name = string("x_201_cast_fp16")]; + tensor var_7067 = const()[name = string("op_7067"), val = tensor([1, 2, 1, 1])]; + tensor x_203_cast_fp16 = tile(reps = var_7067, x = x_201_cast_fp16)[name = string("x_203_cast_fp16")]; + tensor var_7079 = const()[name = string("op_7079"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_75_cast_fp16 = reshape(shape = var_7079, x = x_203_cast_fp16)[name = string("value_states_75_cast_fp16")]; + bool var_7094_transpose_x_1 = const()[name = string("op_7094_transpose_x_1"), val = bool(false)]; + bool var_7094_transpose_y_1 = const()[name = string("op_7094_transpose_y_1"), val = bool(true)]; + tensor var_7094 = matmul(transpose_x = var_7094_transpose_x_1, transpose_y = var_7094_transpose_y_1, x = query_states_49, y = key_states_51_cast_fp16)[name = string("op_7094")]; + fp16 var_7095_to_fp16 = const()[name = string("op_7095_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_73_cast_fp16 = mul(x = var_7094, y = var_7095_to_fp16)[name = string("attn_weights_73_cast_fp16")]; + tensor attn_weights_75_cast_fp16 = add(x = attn_weights_73_cast_fp16, y = causal_mask)[name = string("attn_weights_75_cast_fp16")]; + int32 var_7130 = const()[name = string("op_7130"), val = int32(-1)]; + tensor attn_weights_77_cast_fp16 = softmax(axis = var_7130, x = attn_weights_75_cast_fp16)[name = string("attn_weights_77_cast_fp16")]; + bool attn_output_121_transpose_x_0 = const()[name = string("attn_output_121_transpose_x_0"), val = bool(false)]; + bool attn_output_121_transpose_y_0 = const()[name = string("attn_output_121_transpose_y_0"), val = bool(false)]; + tensor attn_output_121_cast_fp16 = matmul(transpose_x = attn_output_121_transpose_x_0, transpose_y = attn_output_121_transpose_y_0, x = attn_weights_77_cast_fp16, y = value_states_75_cast_fp16)[name = string("attn_output_121_cast_fp16")]; + tensor var_7141_perm_0 = const()[name = string("op_7141_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_7145 = const()[name = string("op_7145"), val = tensor([1, 1, 2048])]; + tensor var_7141_cast_fp16 = transpose(perm = var_7141_perm_0, x = attn_output_121_cast_fp16)[name = string("transpose_10")]; + tensor attn_output_125_cast_fp16 = reshape(shape = var_7145, x = var_7141_cast_fp16)[name = string("attn_output_125_cast_fp16")]; + tensor var_7150 = const()[name = string("op_7150"), val = tensor([0, 2, 1])]; + string var_7166_pad_type_0 = const()[name = string("op_7166_pad_type_0"), val = string("valid")]; + int32 var_7166_groups_0 = const()[name = string("op_7166_groups_0"), val = int32(1)]; + tensor var_7166_strides_0 = const()[name = string("op_7166_strides_0"), val = tensor([1])]; + tensor var_7166_pad_0 = const()[name = string("op_7166_pad_0"), val = tensor([0, 0])]; + tensor var_7166_dilations_0 = const()[name = string("op_7166_dilations_0"), val = tensor([1])]; + tensor squeeze_12_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715515776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719710144))))[name = string("squeeze_12_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_7151_cast_fp16 = transpose(perm = var_7150, x = attn_output_125_cast_fp16)[name = string("transpose_9")]; + tensor var_7166_cast_fp16 = conv(dilations = var_7166_dilations_0, groups = var_7166_groups_0, pad = var_7166_pad_0, pad_type = var_7166_pad_type_0, strides = var_7166_strides_0, weight = squeeze_12_cast_fp16_to_fp32_to_fp16_palettized, x = var_7151_cast_fp16)[name = string("op_7166_cast_fp16")]; + tensor var_7170 = const()[name = string("op_7170"), val = tensor([0, 2, 1])]; + tensor attn_output_129_cast_fp16 = transpose(perm = var_7170, x = var_7166_cast_fp16)[name = string("transpose_8")]; + tensor hidden_states_129_cast_fp16 = add(x = hidden_states_121_cast_fp16, y = attn_output_129_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; + int32 var_7183 = const()[name = string("op_7183"), val = int32(-1)]; + fp16 const_386_promoted_to_fp16 = const()[name = string("const_386_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7185_cast_fp16 = mul(x = hidden_states_129_cast_fp16, y = const_386_promoted_to_fp16)[name = string("op_7185_cast_fp16")]; + bool input_227_interleave_0 = const()[name = string("input_227_interleave_0"), val = bool(false)]; + tensor input_227_cast_fp16 = concat(axis = var_7183, interleave = input_227_interleave_0, values = (hidden_states_129_cast_fp16, var_7185_cast_fp16))[name = string("input_227_cast_fp16")]; + tensor normed_205_axes_0 = const()[name = string("normed_205_axes_0"), val = tensor([-1])]; + fp16 var_7180_to_fp16 = const()[name = string("op_7180_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_205_cast_fp16 = layer_norm(axes = normed_205_axes_0, epsilon = var_7180_to_fp16, x = input_227_cast_fp16)[name = string("normed_205_cast_fp16")]; + tensor normed_207_begin_0 = const()[name = string("normed_207_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_207_end_0 = const()[name = string("normed_207_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_207_end_mask_0 = const()[name = string("normed_207_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_207_cast_fp16 = slice_by_index(begin = normed_207_begin_0, end = normed_207_end_0, end_mask = normed_207_end_mask_0, x = normed_205_cast_fp16)[name = string("normed_207_cast_fp16")]; + tensor const_389_promoted_to_fp16 = const()[name = string("const_389_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719841280)))]; + tensor x_205_cast_fp16 = mul(x = normed_207_cast_fp16, y = const_389_promoted_to_fp16)[name = string("x_205_cast_fp16")]; + tensor var_7210 = const()[name = string("op_7210"), val = tensor([0, 2, 1])]; + tensor input_229_axes_0 = const()[name = string("input_229_axes_0"), val = tensor([2])]; + tensor var_7211 = transpose(perm = var_7210, x = x_205_cast_fp16)[name = string("transpose_7")]; + tensor input_229 = expand_dims(axes = input_229_axes_0, x = var_7211)[name = string("input_229")]; + string input_231_pad_type_0 = const()[name = string("input_231_pad_type_0"), val = string("valid")]; + tensor input_231_strides_0 = const()[name = string("input_231_strides_0"), val = tensor([1, 1])]; + tensor input_231_pad_0 = const()[name = string("input_231_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_231_dilations_0 = const()[name = string("input_231_dilations_0"), val = tensor([1, 1])]; + int32 input_231_groups_0 = const()[name = string("input_231_groups_0"), val = int32(1)]; + tensor input_231 = conv(dilations = input_231_dilations_0, groups = input_231_groups_0, pad = input_231_pad_0, pad_type = input_231_pad_type_0, strides = input_231_strides_0, weight = model_model_layers_12_mlp_gate_proj_weight_palettized, x = input_229)[name = string("input_231")]; + string b_25_pad_type_0 = const()[name = string("b_25_pad_type_0"), val = string("valid")]; + tensor b_25_strides_0 = const()[name = string("b_25_strides_0"), val = tensor([1, 1])]; + tensor b_25_pad_0 = const()[name = string("b_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_25_dilations_0 = const()[name = string("b_25_dilations_0"), val = tensor([1, 1])]; + int32 b_25_groups_0 = const()[name = string("b_25_groups_0"), val = int32(1)]; + tensor b_25 = conv(dilations = b_25_dilations_0, groups = b_25_groups_0, pad = b_25_pad_0, pad_type = b_25_pad_type_0, strides = b_25_strides_0, weight = model_model_layers_12_mlp_up_proj_weight_palettized, x = input_229)[name = string("b_25")]; + tensor c_25 = silu(x = input_231)[name = string("c_25")]; + tensor input_233 = mul(x = c_25, y = b_25)[name = string("input_233")]; + string e_25_pad_type_0 = const()[name = string("e_25_pad_type_0"), val = string("valid")]; + tensor e_25_strides_0 = const()[name = string("e_25_strides_0"), val = tensor([1, 1])]; + tensor e_25_pad_0 = const()[name = string("e_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_25_dilations_0 = const()[name = string("e_25_dilations_0"), val = tensor([1, 1])]; + int32 e_25_groups_0 = const()[name = string("e_25_groups_0"), val = int32(1)]; + tensor e_25 = conv(dilations = e_25_dilations_0, groups = e_25_groups_0, pad = e_25_pad_0, pad_type = e_25_pad_type_0, strides = e_25_strides_0, weight = model_model_layers_12_mlp_down_proj_weight_palettized, x = input_233)[name = string("e_25")]; + tensor var_7233_axes_0 = const()[name = string("op_7233_axes_0"), val = tensor([2])]; + tensor var_7233 = squeeze(axes = var_7233_axes_0, x = e_25)[name = string("op_7233")]; + tensor var_7234 = const()[name = string("op_7234"), val = tensor([0, 2, 1])]; + tensor var_7235 = transpose(perm = var_7234, x = var_7233)[name = string("transpose_6")]; + tensor hidden_states_131_cast_fp16 = add(x = hidden_states_129_cast_fp16, y = var_7235)[name = string("hidden_states_131_cast_fp16")]; + int32 var_7247 = const()[name = string("op_7247"), val = int32(-1)]; + fp16 const_390_promoted_to_fp16 = const()[name = string("const_390_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7249_cast_fp16 = mul(x = hidden_states_131_cast_fp16, y = const_390_promoted_to_fp16)[name = string("op_7249_cast_fp16")]; + bool input_235_interleave_0 = const()[name = string("input_235_interleave_0"), val = bool(false)]; + tensor input_235_cast_fp16 = concat(axis = var_7247, interleave = input_235_interleave_0, values = (hidden_states_131_cast_fp16, var_7249_cast_fp16))[name = string("input_235_cast_fp16")]; + tensor normed_209_axes_0 = const()[name = string("normed_209_axes_0"), val = tensor([-1])]; + fp16 var_7244_to_fp16 = const()[name = string("op_7244_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_209_cast_fp16 = layer_norm(axes = normed_209_axes_0, epsilon = var_7244_to_fp16, x = input_235_cast_fp16)[name = string("normed_209_cast_fp16")]; + tensor normed_211_begin_0 = const()[name = string("normed_211_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_211_end_0 = const()[name = string("normed_211_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_211_end_mask_0 = const()[name = string("normed_211_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_211_cast_fp16 = slice_by_index(begin = normed_211_begin_0, end = normed_211_end_0, end_mask = normed_211_end_mask_0, x = normed_209_cast_fp16)[name = string("normed_211_cast_fp16")]; + tensor const_393_promoted_to_fp16 = const()[name = string("const_393_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719845440)))]; + tensor hidden_states_133_cast_fp16 = mul(x = normed_211_cast_fp16, y = const_393_promoted_to_fp16)[name = string("hidden_states_133_cast_fp16")]; + tensor var_7266 = const()[name = string("op_7266"), val = tensor([0, 2, 1])]; + tensor var_7269_axes_0 = const()[name = string("op_7269_axes_0"), val = tensor([2])]; + tensor var_7267_cast_fp16 = transpose(perm = var_7266, x = hidden_states_133_cast_fp16)[name = string("transpose_5")]; + tensor var_7269_cast_fp16 = expand_dims(axes = var_7269_axes_0, x = var_7267_cast_fp16)[name = string("op_7269_cast_fp16")]; + string var_7285_pad_type_0 = const()[name = string("op_7285_pad_type_0"), val = string("valid")]; + tensor var_7285_strides_0 = const()[name = string("op_7285_strides_0"), val = tensor([1, 1])]; + tensor var_7285_pad_0 = const()[name = string("op_7285_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7285_dilations_0 = const()[name = string("op_7285_dilations_0"), val = tensor([1, 1])]; + int32 var_7285_groups_0 = const()[name = string("op_7285_groups_0"), val = int32(1)]; + tensor var_7285 = conv(dilations = var_7285_dilations_0, groups = var_7285_groups_0, pad = var_7285_pad_0, pad_type = var_7285_pad_type_0, strides = var_7285_strides_0, weight = model_model_layers_13_self_attn_q_proj_weight_palettized, x = var_7269_cast_fp16)[name = string("op_7285")]; + tensor var_7290 = const()[name = string("op_7290"), val = tensor([1, 16, 1, 128])]; + tensor var_7291 = reshape(shape = var_7290, x = var_7285)[name = string("op_7291")]; + string var_7307_pad_type_0 = const()[name = string("op_7307_pad_type_0"), val = string("valid")]; + tensor var_7307_strides_0 = const()[name = string("op_7307_strides_0"), val = tensor([1, 1])]; + tensor var_7307_pad_0 = const()[name = string("op_7307_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7307_dilations_0 = const()[name = string("op_7307_dilations_0"), val = tensor([1, 1])]; + int32 var_7307_groups_0 = const()[name = string("op_7307_groups_0"), val = int32(1)]; + tensor var_7307 = conv(dilations = var_7307_dilations_0, groups = var_7307_groups_0, pad = var_7307_pad_0, pad_type = var_7307_pad_type_0, strides = var_7307_strides_0, weight = model_model_layers_13_self_attn_k_proj_weight_palettized, x = var_7269_cast_fp16)[name = string("op_7307")]; + tensor var_7312 = const()[name = string("op_7312"), val = tensor([1, 8, 1, 128])]; + tensor var_7313 = reshape(shape = var_7312, x = var_7307)[name = string("op_7313")]; + string var_7329_pad_type_0 = const()[name = string("op_7329_pad_type_0"), val = string("valid")]; + tensor var_7329_strides_0 = const()[name = string("op_7329_strides_0"), val = tensor([1, 1])]; + tensor var_7329_pad_0 = const()[name = string("op_7329_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7329_dilations_0 = const()[name = string("op_7329_dilations_0"), val = tensor([1, 1])]; + int32 var_7329_groups_0 = const()[name = string("op_7329_groups_0"), val = int32(1)]; + tensor var_7329 = conv(dilations = var_7329_dilations_0, groups = var_7329_groups_0, pad = var_7329_pad_0, pad_type = var_7329_pad_type_0, strides = var_7329_strides_0, weight = model_model_layers_13_self_attn_v_proj_weight_palettized, x = var_7269_cast_fp16)[name = string("op_7329")]; + tensor var_7334 = const()[name = string("op_7334"), val = tensor([1, 8, 1, 128])]; + tensor var_7335 = reshape(shape = var_7334, x = var_7329)[name = string("op_7335")]; + int32 var_7350 = const()[name = string("op_7350"), val = int32(-1)]; + fp16 const_394_promoted = const()[name = string("const_394_promoted"), val = fp16(-0x1p+0)]; + tensor var_7352 = mul(x = var_7291, y = const_394_promoted)[name = string("op_7352")]; + bool input_239_interleave_0 = const()[name = string("input_239_interleave_0"), val = bool(false)]; + tensor input_239 = concat(axis = var_7350, interleave = input_239_interleave_0, values = (var_7291, var_7352))[name = string("input_239")]; + tensor normed_213_axes_0 = const()[name = string("normed_213_axes_0"), val = tensor([-1])]; + fp16 var_7347_to_fp16 = const()[name = string("op_7347_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_213_cast_fp16 = layer_norm(axes = normed_213_axes_0, epsilon = var_7347_to_fp16, x = input_239)[name = string("normed_213_cast_fp16")]; + tensor normed_215_begin_0 = const()[name = string("normed_215_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_215_end_0 = const()[name = string("normed_215_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_215_end_mask_0 = const()[name = string("normed_215_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_215 = slice_by_index(begin = normed_215_begin_0, end = normed_215_end_0, end_mask = normed_215_end_mask_0, x = normed_213_cast_fp16)[name = string("normed_215")]; + tensor const_397 = const()[name = string("const_397"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719849600)))]; + tensor q = mul(x = normed_215, y = const_397)[name = string("q")]; + int32 var_7375 = const()[name = string("op_7375"), val = int32(-1)]; + fp16 const_398_promoted = const()[name = string("const_398_promoted"), val = fp16(-0x1p+0)]; + tensor var_7377 = mul(x = var_7313, y = const_398_promoted)[name = string("op_7377")]; + bool input_241_interleave_0 = const()[name = string("input_241_interleave_0"), val = bool(false)]; + tensor input_241 = concat(axis = var_7375, interleave = input_241_interleave_0, values = (var_7313, var_7377))[name = string("input_241")]; + tensor normed_217_axes_0 = const()[name = string("normed_217_axes_0"), val = tensor([-1])]; + fp16 var_7372_to_fp16 = const()[name = string("op_7372_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_217_cast_fp16 = layer_norm(axes = normed_217_axes_0, epsilon = var_7372_to_fp16, x = input_241)[name = string("normed_217_cast_fp16")]; + tensor normed_219_begin_0 = const()[name = string("normed_219_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_219_end_0 = const()[name = string("normed_219_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_219_end_mask_0 = const()[name = string("normed_219_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_219 = slice_by_index(begin = normed_219_begin_0, end = normed_219_end_0, end_mask = normed_219_end_mask_0, x = normed_217_cast_fp16)[name = string("normed_219")]; + tensor const_401 = const()[name = string("const_401"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719849920)))]; + tensor k = mul(x = normed_219, y = const_401)[name = string("k")]; + tensor var_7391 = mul(x = q, y = cos_1_cast_fp16)[name = string("op_7391")]; + tensor x1_53_begin_0 = const()[name = string("x1_53_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_53_end_0 = const()[name = string("x1_53_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_53_end_mask_0 = const()[name = string("x1_53_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_53 = slice_by_index(begin = x1_53_begin_0, end = x1_53_end_0, end_mask = x1_53_end_mask_0, x = q)[name = string("x1_53")]; + tensor x2_53_begin_0 = const()[name = string("x2_53_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_53_end_0 = const()[name = string("x2_53_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_53_end_mask_0 = const()[name = string("x2_53_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_53 = slice_by_index(begin = x2_53_begin_0, end = x2_53_end_0, end_mask = x2_53_end_mask_0, x = q)[name = string("x2_53")]; + fp16 const_404_promoted = const()[name = string("const_404_promoted"), val = fp16(-0x1p+0)]; + tensor var_7412 = mul(x = x2_53, y = const_404_promoted)[name = string("op_7412")]; + int32 var_7414 = const()[name = string("op_7414"), val = int32(-1)]; + bool var_7415_interleave_0 = const()[name = string("op_7415_interleave_0"), val = bool(false)]; + tensor var_7415 = concat(axis = var_7414, interleave = var_7415_interleave_0, values = (var_7412, x1_53))[name = string("op_7415")]; + tensor var_7416 = mul(x = var_7415, y = sin_1_cast_fp16)[name = string("op_7416")]; + tensor query_states_53 = add(x = var_7391, y = var_7416)[name = string("query_states_53")]; + tensor var_7419 = mul(x = k, y = cos_1_cast_fp16)[name = string("op_7419")]; + tensor x1_begin_0 = const()[name = string("x1_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_end_0 = const()[name = string("x1_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_end_mask_0 = const()[name = string("x1_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1 = slice_by_index(begin = x1_begin_0, end = x1_end_0, end_mask = x1_end_mask_0, x = k)[name = string("x1")]; + tensor x2_begin_0 = const()[name = string("x2_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_end_0 = const()[name = string("x2_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_end_mask_0 = const()[name = string("x2_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2 = slice_by_index(begin = x2_begin_0, end = x2_end_0, end_mask = x2_end_mask_0, x = k)[name = string("x2")]; + fp16 const_407_promoted = const()[name = string("const_407_promoted"), val = fp16(-0x1p+0)]; + tensor var_7440 = mul(x = x2, y = const_407_promoted)[name = string("op_7440")]; + int32 var_7442 = const()[name = string("op_7442"), val = int32(-1)]; + bool var_7443_interleave_0 = const()[name = string("op_7443_interleave_0"), val = bool(false)]; + tensor var_7443 = concat(axis = var_7442, interleave = var_7443_interleave_0, values = (var_7440, x1))[name = string("op_7443")]; + tensor var_7444 = mul(x = var_7443, y = sin_1_cast_fp16)[name = string("op_7444")]; + tensor key_states_53 = add(x = var_7419, y = var_7444)[name = string("key_states_53")]; + tensor expand_dims_156 = const()[name = string("expand_dims_156"), val = tensor([13])]; + tensor expand_dims_157 = const()[name = string("expand_dims_157"), val = tensor([0])]; + tensor expand_dims_159 = const()[name = string("expand_dims_159"), val = tensor([0])]; + tensor expand_dims_160 = const()[name = string("expand_dims_160"), val = tensor([14])]; + int32 concat_106_axis_0 = const()[name = string("concat_106_axis_0"), val = int32(0)]; + bool concat_106_interleave_0 = const()[name = string("concat_106_interleave_0"), val = bool(false)]; + tensor concat_106 = concat(axis = concat_106_axis_0, interleave = concat_106_interleave_0, values = (expand_dims_156, expand_dims_157, current_pos, expand_dims_159))[name = string("concat_106")]; + tensor concat_107_values1_0 = const()[name = string("concat_107_values1_0"), val = tensor([0])]; + tensor concat_107_values3_0 = const()[name = string("concat_107_values3_0"), val = tensor([0])]; + int32 concat_107_axis_0 = const()[name = string("concat_107_axis_0"), val = int32(0)]; + bool concat_107_interleave_0 = const()[name = string("concat_107_interleave_0"), val = bool(false)]; + tensor concat_107 = concat(axis = concat_107_axis_0, interleave = concat_107_interleave_0, values = (expand_dims_160, concat_107_values1_0, var_1001, concat_107_values3_0))[name = string("concat_107")]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_106, begin_mask = model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0, end = concat_107, end_mask = model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_27_stride_0, update = key_states_53, x = coreml_update_state_53)[name = string("model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_82_write_state")]; + tensor coreml_update_state_54 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_82")]; + tensor expand_dims_162 = const()[name = string("expand_dims_162"), val = tensor([41])]; + tensor expand_dims_163 = const()[name = string("expand_dims_163"), val = tensor([0])]; + tensor expand_dims_165 = const()[name = string("expand_dims_165"), val = tensor([0])]; + tensor expand_dims_166 = const()[name = string("expand_dims_166"), val = tensor([42])]; + int32 concat_110_axis_0 = const()[name = string("concat_110_axis_0"), val = int32(0)]; + bool concat_110_interleave_0 = const()[name = string("concat_110_interleave_0"), val = bool(false)]; + tensor concat_110 = concat(axis = concat_110_axis_0, interleave = concat_110_interleave_0, values = (expand_dims_162, expand_dims_163, current_pos, expand_dims_165))[name = string("concat_110")]; + tensor concat_111_values1_0 = const()[name = string("concat_111_values1_0"), val = tensor([0])]; + tensor concat_111_values3_0 = const()[name = string("concat_111_values3_0"), val = tensor([0])]; + int32 concat_111_axis_0 = const()[name = string("concat_111_axis_0"), val = int32(0)]; + bool concat_111_interleave_0 = const()[name = string("concat_111_interleave_0"), val = bool(false)]; + tensor concat_111 = concat(axis = concat_111_axis_0, interleave = concat_111_interleave_0, values = (expand_dims_166, concat_111_values1_0, var_1001, concat_111_values3_0))[name = string("concat_111")]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_110, begin_mask = model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0, end = concat_111, end_mask = model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_28_stride_0, update = var_7335, x = coreml_update_state_54)[name = string("model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_83_write_state")]; + tensor coreml_update_state_55 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_83")]; + tensor var_7499_begin_0 = const()[name = string("op_7499_begin_0"), val = tensor([13, 0, 0, 0])]; + tensor var_7499_end_0 = const()[name = string("op_7499_end_0"), val = tensor([14, 8, 1024, 128])]; + tensor var_7499_end_mask_0 = const()[name = string("op_7499_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_7499_cast_fp16 = slice_by_index(begin = var_7499_begin_0, end = var_7499_end_0, end_mask = var_7499_end_mask_0, x = coreml_update_state_55)[name = string("op_7499_cast_fp16")]; + tensor K_layer_cache_axes_0 = const()[name = string("K_layer_cache_axes_0"), val = tensor([0])]; + tensor K_layer_cache_cast_fp16 = squeeze(axes = K_layer_cache_axes_0, x = var_7499_cast_fp16)[name = string("K_layer_cache_cast_fp16")]; + tensor var_7506_begin_0 = const()[name = string("op_7506_begin_0"), val = tensor([41, 0, 0, 0])]; + tensor var_7506_end_0 = const()[name = string("op_7506_end_0"), val = tensor([42, 8, 1024, 128])]; + tensor var_7506_end_mask_0 = const()[name = string("op_7506_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_7506_cast_fp16 = slice_by_index(begin = var_7506_begin_0, end = var_7506_end_0, end_mask = var_7506_end_mask_0, x = coreml_update_state_55)[name = string("op_7506_cast_fp16")]; + tensor V_layer_cache_axes_0 = const()[name = string("V_layer_cache_axes_0"), val = tensor([0])]; + tensor V_layer_cache_cast_fp16 = squeeze(axes = V_layer_cache_axes_0, x = var_7506_cast_fp16)[name = string("V_layer_cache_cast_fp16")]; + tensor x_211_axes_0 = const()[name = string("x_211_axes_0"), val = tensor([1])]; + tensor x_211_cast_fp16 = expand_dims(axes = x_211_axes_0, x = K_layer_cache_cast_fp16)[name = string("x_211_cast_fp16")]; + tensor var_7543 = const()[name = string("op_7543"), val = tensor([1, 2, 1, 1])]; + tensor x_213_cast_fp16 = tile(reps = var_7543, x = x_211_cast_fp16)[name = string("x_213_cast_fp16")]; + tensor var_7555 = const()[name = string("op_7555"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_cast_fp16 = reshape(shape = var_7555, x = x_213_cast_fp16)[name = string("key_states_cast_fp16")]; + tensor x_217_axes_0 = const()[name = string("x_217_axes_0"), val = tensor([1])]; + tensor x_217_cast_fp16 = expand_dims(axes = x_217_axes_0, x = V_layer_cache_cast_fp16)[name = string("x_217_cast_fp16")]; + tensor var_7563 = const()[name = string("op_7563"), val = tensor([1, 2, 1, 1])]; + tensor x_219_cast_fp16 = tile(reps = var_7563, x = x_217_cast_fp16)[name = string("x_219_cast_fp16")]; + tensor var_7575 = const()[name = string("op_7575"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_81_cast_fp16 = reshape(shape = var_7575, x = x_219_cast_fp16)[name = string("value_states_81_cast_fp16")]; + bool var_7590_transpose_x_1 = const()[name = string("op_7590_transpose_x_1"), val = bool(false)]; + bool var_7590_transpose_y_1 = const()[name = string("op_7590_transpose_y_1"), val = bool(true)]; + tensor var_7590 = matmul(transpose_x = var_7590_transpose_x_1, transpose_y = var_7590_transpose_y_1, x = query_states_53, y = key_states_cast_fp16)[name = string("op_7590")]; + fp16 var_7591_to_fp16 = const()[name = string("op_7591_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_79_cast_fp16 = mul(x = var_7590, y = var_7591_to_fp16)[name = string("attn_weights_79_cast_fp16")]; + tensor attn_weights_81_cast_fp16 = add(x = attn_weights_79_cast_fp16, y = causal_mask)[name = string("attn_weights_81_cast_fp16")]; + int32 var_7626 = const()[name = string("op_7626"), val = int32(-1)]; + tensor attn_weights_cast_fp16 = softmax(axis = var_7626, x = attn_weights_81_cast_fp16)[name = string("attn_weights_cast_fp16")]; + bool attn_output_131_transpose_x_0 = const()[name = string("attn_output_131_transpose_x_0"), val = bool(false)]; + bool attn_output_131_transpose_y_0 = const()[name = string("attn_output_131_transpose_y_0"), val = bool(false)]; + tensor attn_output_131_cast_fp16 = matmul(transpose_x = attn_output_131_transpose_x_0, transpose_y = attn_output_131_transpose_y_0, x = attn_weights_cast_fp16, y = value_states_81_cast_fp16)[name = string("attn_output_131_cast_fp16")]; + tensor var_7637_perm_0 = const()[name = string("op_7637_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_7641 = const()[name = string("op_7641"), val = tensor([1, 1, 2048])]; + tensor var_7637_cast_fp16 = transpose(perm = var_7637_perm_0, x = attn_output_131_cast_fp16)[name = string("transpose_4")]; + tensor attn_output_135_cast_fp16 = reshape(shape = var_7641, x = var_7637_cast_fp16)[name = string("attn_output_135_cast_fp16")]; + tensor var_7646 = const()[name = string("op_7646"), val = tensor([0, 2, 1])]; + string var_7662_pad_type_0 = const()[name = string("op_7662_pad_type_0"), val = string("valid")]; + int32 var_7662_groups_0 = const()[name = string("op_7662_groups_0"), val = int32(1)]; + tensor var_7662_strides_0 = const()[name = string("op_7662_strides_0"), val = tensor([1])]; + tensor var_7662_pad_0 = const()[name = string("op_7662_pad_0"), val = tensor([0, 0])]; + tensor var_7662_dilations_0 = const()[name = string("op_7662_dilations_0"), val = tensor([1])]; + tensor squeeze_13_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719850240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724044608))))[name = string("squeeze_13_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_7647_cast_fp16 = transpose(perm = var_7646, x = attn_output_135_cast_fp16)[name = string("transpose_3")]; + tensor var_7662_cast_fp16 = conv(dilations = var_7662_dilations_0, groups = var_7662_groups_0, pad = var_7662_pad_0, pad_type = var_7662_pad_type_0, strides = var_7662_strides_0, weight = squeeze_13_cast_fp16_to_fp32_to_fp16_palettized, x = var_7647_cast_fp16)[name = string("op_7662_cast_fp16")]; + tensor var_7666 = const()[name = string("op_7666"), val = tensor([0, 2, 1])]; + tensor attn_output_cast_fp16 = transpose(perm = var_7666, x = var_7662_cast_fp16)[name = string("transpose_2")]; + tensor hidden_states_cast_fp16 = add(x = hidden_states_131_cast_fp16, y = attn_output_cast_fp16)[name = string("hidden_states_cast_fp16")]; + int32 var_7679 = const()[name = string("op_7679"), val = int32(-1)]; + fp16 const_416_promoted_to_fp16 = const()[name = string("const_416_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7681_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_416_promoted_to_fp16)[name = string("op_7681_cast_fp16")]; + bool input_245_interleave_0 = const()[name = string("input_245_interleave_0"), val = bool(false)]; + tensor input_245_cast_fp16 = concat(axis = var_7679, interleave = input_245_interleave_0, values = (hidden_states_cast_fp16, var_7681_cast_fp16))[name = string("input_245_cast_fp16")]; + tensor normed_221_axes_0 = const()[name = string("normed_221_axes_0"), val = tensor([-1])]; + fp16 var_7676_to_fp16 = const()[name = string("op_7676_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_221_cast_fp16 = layer_norm(axes = normed_221_axes_0, epsilon = var_7676_to_fp16, x = input_245_cast_fp16)[name = string("normed_221_cast_fp16")]; + tensor normed_begin_0 = const()[name = string("normed_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_end_0 = const()[name = string("normed_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_end_mask_0 = const()[name = string("normed_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_cast_fp16 = slice_by_index(begin = normed_begin_0, end = normed_end_0, end_mask = normed_end_mask_0, x = normed_221_cast_fp16)[name = string("normed_cast_fp16")]; + tensor const_419_promoted_to_fp16 = const()[name = string("const_419_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724175744)))]; + tensor x_221_cast_fp16 = mul(x = normed_cast_fp16, y = const_419_promoted_to_fp16)[name = string("x_221_cast_fp16")]; + tensor var_7706 = const()[name = string("op_7706"), val = tensor([0, 2, 1])]; + tensor input_247_axes_0 = const()[name = string("input_247_axes_0"), val = tensor([2])]; + tensor var_7707 = transpose(perm = var_7706, x = x_221_cast_fp16)[name = string("transpose_1")]; + tensor input_247 = expand_dims(axes = input_247_axes_0, x = var_7707)[name = string("input_247")]; + string input_249_pad_type_0 = const()[name = string("input_249_pad_type_0"), val = string("valid")]; + tensor input_249_strides_0 = const()[name = string("input_249_strides_0"), val = tensor([1, 1])]; + tensor input_249_pad_0 = const()[name = string("input_249_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_249_dilations_0 = const()[name = string("input_249_dilations_0"), val = tensor([1, 1])]; + int32 input_249_groups_0 = const()[name = string("input_249_groups_0"), val = int32(1)]; + tensor input_249 = conv(dilations = input_249_dilations_0, groups = input_249_groups_0, pad = input_249_pad_0, pad_type = input_249_pad_type_0, strides = input_249_strides_0, weight = model_model_layers_13_mlp_gate_proj_weight_palettized, x = input_247)[name = string("input_249")]; + string b_pad_type_0 = const()[name = string("b_pad_type_0"), val = string("valid")]; + tensor b_strides_0 = const()[name = string("b_strides_0"), val = tensor([1, 1])]; + tensor b_pad_0 = const()[name = string("b_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_dilations_0 = const()[name = string("b_dilations_0"), val = tensor([1, 1])]; + int32 b_groups_0 = const()[name = string("b_groups_0"), val = int32(1)]; + tensor b = conv(dilations = b_dilations_0, groups = b_groups_0, pad = b_pad_0, pad_type = b_pad_type_0, strides = b_strides_0, weight = model_model_layers_13_mlp_up_proj_weight_palettized, x = input_247)[name = string("b")]; + tensor c = silu(x = input_249)[name = string("c")]; + tensor input = mul(x = c, y = b)[name = string("input")]; + string e_pad_type_0 = const()[name = string("e_pad_type_0"), val = string("valid")]; + tensor e_strides_0 = const()[name = string("e_strides_0"), val = tensor([1, 1])]; + tensor e_pad_0 = const()[name = string("e_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_dilations_0 = const()[name = string("e_dilations_0"), val = tensor([1, 1])]; + int32 e_groups_0 = const()[name = string("e_groups_0"), val = int32(1)]; + tensor e = conv(dilations = e_dilations_0, groups = e_groups_0, pad = e_pad_0, pad_type = e_pad_type_0, strides = e_strides_0, weight = model_model_layers_13_mlp_down_proj_weight_palettized, x = input)[name = string("e")]; + tensor var_7729_axes_0 = const()[name = string("op_7729_axes_0"), val = tensor([2])]; + tensor var_7729 = squeeze(axes = var_7729_axes_0, x = e)[name = string("op_7729")]; + tensor var_7730 = const()[name = string("op_7730"), val = tensor([0, 2, 1])]; + tensor var_7731 = transpose(perm = var_7730, x = var_7729)[name = string("transpose_0")]; + tensor output_hidden_states = add(x = hidden_states_cast_fp16, y = var_7731)[name = string("op_7733_cast_fp16")]; + tensor position_ids_tmp = identity(x = position_ids)[name = string("position_ids_tmp")]; + } -> (output_hidden_states); + func prefill(tensor causal_mask, tensor current_pos, tensor hidden_states, state> model_model_kv_cache_0, tensor position_ids) { + tensor model_model_layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4194432))))[name = string("model_model_layers_0_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_0_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4325568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6422784))))[name = string("model_model_layers_0_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_0_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6488384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8585600))))[name = string("model_model_layers_0_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_0_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8651200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21234176))))[name = string("model_model_layers_0_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_0_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21627456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34210432))))[name = string("model_model_layers_0_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_0_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34603712))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47186688))))[name = string("model_model_layers_0_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_1_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47317824))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51512192))))[name = string("model_model_layers_1_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_1_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51643328))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53740544))))[name = string("model_model_layers_1_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_1_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53806144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55903360))))[name = string("model_model_layers_1_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_1_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55968960))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68551936))))[name = string("model_model_layers_1_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_1_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68945216))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(81528192))))[name = string("model_model_layers_1_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_1_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(81921472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(94504448))))[name = string("model_model_layers_1_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(94635584))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(98829952))))[name = string("model_model_layers_2_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_2_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(98961088))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101058304))))[name = string("model_model_layers_2_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_2_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101123904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103221120))))[name = string("model_model_layers_2_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_2_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103286720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115869696))))[name = string("model_model_layers_2_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_2_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116262976))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(128845952))))[name = string("model_model_layers_2_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_2_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(129239232))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141822208))))[name = string("model_model_layers_2_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_3_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141953344))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146147712))))[name = string("model_model_layers_3_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_3_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146278848))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148376064))))[name = string("model_model_layers_3_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_3_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148441664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(150538880))))[name = string("model_model_layers_3_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_3_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(150604480))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163187456))))[name = string("model_model_layers_3_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_3_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163580736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176163712))))[name = string("model_model_layers_3_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_3_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176556992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(189139968))))[name = string("model_model_layers_3_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_4_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(189271104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193465472))))[name = string("model_model_layers_4_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_4_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193596608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(195693824))))[name = string("model_model_layers_4_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_4_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(195759424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(197856640))))[name = string("model_model_layers_4_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_4_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(197922240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210505216))))[name = string("model_model_layers_4_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_4_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210898496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223481472))))[name = string("model_model_layers_4_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_4_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223874752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236457728))))[name = string("model_model_layers_4_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_5_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236588864))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240783232))))[name = string("model_model_layers_5_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_5_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240914368))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(243011584))))[name = string("model_model_layers_5_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_5_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(243077184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245174400))))[name = string("model_model_layers_5_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_5_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245240000))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(257822976))))[name = string("model_model_layers_5_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_5_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258216256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(270799232))))[name = string("model_model_layers_5_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_5_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(271192512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283775488))))[name = string("model_model_layers_5_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_6_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283906624))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288100992))))[name = string("model_model_layers_6_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_6_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288232128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290329344))))[name = string("model_model_layers_6_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_6_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290394944))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292492160))))[name = string("model_model_layers_6_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_6_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292557760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305140736))))[name = string("model_model_layers_6_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_6_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305534016))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318116992))))[name = string("model_model_layers_6_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_6_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318510272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331093248))))[name = string("model_model_layers_6_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_7_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331224384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335418752))))[name = string("model_model_layers_7_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_7_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335549888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(337647104))))[name = string("model_model_layers_7_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_7_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(337712704))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339809920))))[name = string("model_model_layers_7_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_7_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339875520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352458496))))[name = string("model_model_layers_7_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_7_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352851776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(365434752))))[name = string("model_model_layers_7_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_7_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(365828032))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(378411008))))[name = string("model_model_layers_7_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_8_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(378542144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(382736512))))[name = string("model_model_layers_8_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_8_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(382867648))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(384964864))))[name = string("model_model_layers_8_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_8_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385030464))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(387127680))))[name = string("model_model_layers_8_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_8_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(387193280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399776256))))[name = string("model_model_layers_8_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_8_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400169536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(412752512))))[name = string("model_model_layers_8_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_8_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(413145792))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(425728768))))[name = string("model_model_layers_8_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_9_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(425859904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(430054272))))[name = string("model_model_layers_9_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_9_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(430185408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432282624))))[name = string("model_model_layers_9_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_9_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432348224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(434445440))))[name = string("model_model_layers_9_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_9_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(434511040))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(447094016))))[name = string("model_model_layers_9_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_9_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(447487296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(460070272))))[name = string("model_model_layers_9_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_9_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(460463552))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(473046528))))[name = string("model_model_layers_9_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_10_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(473177664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(477372032))))[name = string("model_model_layers_10_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_10_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(477503168))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479600384))))[name = string("model_model_layers_10_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_10_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479665984))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(481763200))))[name = string("model_model_layers_10_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_10_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(481828800))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(494411776))))[name = string("model_model_layers_10_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_10_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(494805056))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(507388032))))[name = string("model_model_layers_10_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_10_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(507781312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(520364288))))[name = string("model_model_layers_10_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_11_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(520495424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524689792))))[name = string("model_model_layers_11_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_11_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524820928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526918144))))[name = string("model_model_layers_11_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_11_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526983744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(529080960))))[name = string("model_model_layers_11_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_11_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(529146560))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(541729536))))[name = string("model_model_layers_11_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_11_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(542122816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554705792))))[name = string("model_model_layers_11_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_11_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(555099072))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(567682048))))[name = string("model_model_layers_11_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_12_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(567813184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(572007552))))[name = string("model_model_layers_12_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_12_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(572138688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(574235904))))[name = string("model_model_layers_12_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_12_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(574301504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(576398720))))[name = string("model_model_layers_12_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_12_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(576464320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(589047296))))[name = string("model_model_layers_12_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_12_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(589440576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(602023552))))[name = string("model_model_layers_12_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_12_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(602416832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(614999808))))[name = string("model_model_layers_12_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_13_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(615130944))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(619325312))))[name = string("model_model_layers_13_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_13_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(619456448))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(621553664))))[name = string("model_model_layers_13_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_13_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(621619264))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(623716480))))[name = string("model_model_layers_13_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_13_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(623782080))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(636365056))))[name = string("model_model_layers_13_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_13_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(636758336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(649341312))))[name = string("model_model_layers_13_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_13_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(649734592))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(662317568))))[name = string("model_model_layers_13_mlp_down_proj_weight_palettized")]; + int32 var_763_batch_dims_0 = const()[name = string("op_763_batch_dims_0"), val = int32(0)]; + bool var_763_validate_indices_0 = const()[name = string("op_763_validate_indices_0"), val = bool(false)]; + tensor var_755_to_fp16 = const()[name = string("op_755_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(662973056)))]; + string position_ids_to_int16_dtype_0 = const()[name = string("position_ids_to_int16_dtype_0"), val = string("int16")]; + string cast_118_dtype_0 = const()[name = string("cast_118_dtype_0"), val = string("int32")]; + int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; + tensor position_ids_to_int16 = cast(dtype = position_ids_to_int16_dtype_0, x = position_ids)[name = string("cast_5")]; + tensor cast_118 = cast(dtype = cast_118_dtype_0, x = position_ids_to_int16)[name = string("cast_4")]; + tensor greater_equal_0 = greater_equal(x = cast_118, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; + int32 slice_by_index_112 = const()[name = string("slice_by_index_112"), val = int32(2048)]; + tensor add_0 = add(x = cast_118, y = slice_by_index_112)[name = string("add_0")]; + tensor select_0 = select(a = cast_118, b = add_0, cond = greater_equal_0)[name = string("select_0")]; + string select_0_to_int16_dtype_0 = const()[name = string("select_0_to_int16_dtype_0"), val = string("int16")]; + string cast_0_dtype_0 = const()[name = string("cast_0_dtype_0"), val = string("int32")]; + int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; + tensor select_0_to_int16 = cast(dtype = select_0_to_int16_dtype_0, x = select_0)[name = string("cast_3")]; + tensor cast_0 = cast(dtype = cast_0_dtype_0, x = select_0_to_int16)[name = string("cast_2")]; + tensor greater_equal_0_1 = greater_equal(x = cast_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; + int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(2048)]; + tensor add_0_1 = add(x = cast_0, y = slice_by_index_0)[name = string("add_0_1")]; + tensor select_0_1 = select(a = cast_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; + int32 op_763_cast_fp16_cast_uint16_cast_uint16_axis_0 = const()[name = string("op_763_cast_fp16_cast_uint16_cast_uint16_axis_0"), val = int32(1)]; + tensor op_763_cast_fp16_cast_uint16_cast_uint16 = gather(axis = op_763_cast_fp16_cast_uint16_cast_uint16_axis_0, batch_dims = var_763_batch_dims_0, indices = select_0_1, validate_indices = var_763_validate_indices_0, x = var_755_to_fp16)[name = string("op_763_cast_fp16_cast_uint16_cast_uint16")]; + tensor var_767 = const()[name = string("op_767"), val = tensor([1, 128, 1, 128])]; + tensor cos_1_cast_fp16 = reshape(shape = var_767, x = op_763_cast_fp16_cast_uint16_cast_uint16)[name = string("cos_1_cast_fp16")]; + int32 var_777_axis_0 = const()[name = string("op_777_axis_0"), val = int32(1)]; + int32 var_777_batch_dims_0 = const()[name = string("op_777_batch_dims_0"), val = int32(0)]; + bool var_777_validate_indices_0 = const()[name = string("op_777_validate_indices_0"), val = bool(false)]; + tensor var_769_to_fp16 = const()[name = string("op_769_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(662448704)))]; + string position_ids_to_uint16_dtype_0 = const()[name = string("position_ids_to_uint16_dtype_0"), val = string("uint16")]; + tensor position_ids_to_uint16 = cast(dtype = position_ids_to_uint16_dtype_0, x = position_ids)[name = string("cast_1")]; + tensor var_777_cast_fp16_cast_uint16 = gather(axis = var_777_axis_0, batch_dims = var_777_batch_dims_0, indices = position_ids_to_uint16, validate_indices = var_777_validate_indices_0, x = var_769_to_fp16)[name = string("op_777_cast_fp16_cast_uint16")]; + tensor var_781 = const()[name = string("op_781"), val = tensor([1, 128, 1, 128])]; + tensor sin_1_cast_fp16 = reshape(shape = var_781, x = var_777_cast_fp16_cast_uint16)[name = string("sin_1_cast_fp16")]; + int32 var_802 = const()[name = string("op_802"), val = int32(-1)]; + fp16 const_1_promoted_to_fp16 = const()[name = string("const_1_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_804_cast_fp16 = mul(x = hidden_states, y = const_1_promoted_to_fp16)[name = string("op_804_cast_fp16")]; + bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; + tensor input_1_cast_fp16 = concat(axis = var_802, interleave = input_1_interleave_0, values = (hidden_states, var_804_cast_fp16))[name = string("input_1_cast_fp16")]; + tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; + fp16 var_799_to_fp16 = const()[name = string("op_799_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_799_to_fp16, x = input_1_cast_fp16)[name = string("normed_1_cast_fp16")]; + tensor normed_3_begin_0 = const()[name = string("normed_3_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_3_end_0 = const()[name = string("normed_3_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_3_end_mask_0 = const()[name = string("normed_3_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_3_cast_fp16 = slice_by_index(begin = normed_3_begin_0, end = normed_3_end_0, end_mask = normed_3_end_mask_0, x = normed_1_cast_fp16)[name = string("normed_3_cast_fp16")]; + tensor const_4_promoted_to_fp16 = const()[name = string("const_4_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(663497408)))]; + tensor hidden_states_3_cast_fp16 = mul(x = normed_3_cast_fp16, y = const_4_promoted_to_fp16)[name = string("hidden_states_3_cast_fp16")]; + tensor var_827 = const()[name = string("op_827"), val = tensor([0, 2, 1])]; + tensor var_830_axes_0 = const()[name = string("op_830_axes_0"), val = tensor([2])]; + tensor var_828_cast_fp16 = transpose(perm = var_827, x = hidden_states_3_cast_fp16)[name = string("transpose_127")]; + tensor var_830_cast_fp16 = expand_dims(axes = var_830_axes_0, x = var_828_cast_fp16)[name = string("op_830_cast_fp16")]; + string query_states_1_pad_type_0 = const()[name = string("query_states_1_pad_type_0"), val = string("valid")]; + tensor query_states_1_strides_0 = const()[name = string("query_states_1_strides_0"), val = tensor([1, 1])]; + tensor query_states_1_pad_0 = const()[name = string("query_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_1_dilations_0 = const()[name = string("query_states_1_dilations_0"), val = tensor([1, 1])]; + int32 query_states_1_groups_0 = const()[name = string("query_states_1_groups_0"), val = int32(1)]; + tensor query_states_1 = conv(dilations = query_states_1_dilations_0, groups = query_states_1_groups_0, pad = query_states_1_pad_0, pad_type = query_states_1_pad_type_0, strides = query_states_1_strides_0, weight = model_model_layers_0_self_attn_q_proj_weight_palettized, x = var_830_cast_fp16)[name = string("query_states_1")]; + string key_states_1_pad_type_0 = const()[name = string("key_states_1_pad_type_0"), val = string("valid")]; + tensor key_states_1_strides_0 = const()[name = string("key_states_1_strides_0"), val = tensor([1, 1])]; + tensor key_states_1_pad_0 = const()[name = string("key_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_1_dilations_0 = const()[name = string("key_states_1_dilations_0"), val = tensor([1, 1])]; + int32 key_states_1_groups_0 = const()[name = string("key_states_1_groups_0"), val = int32(1)]; + tensor key_states_1 = conv(dilations = key_states_1_dilations_0, groups = key_states_1_groups_0, pad = key_states_1_pad_0, pad_type = key_states_1_pad_type_0, strides = key_states_1_strides_0, weight = model_model_layers_0_self_attn_k_proj_weight_palettized, x = var_830_cast_fp16)[name = string("key_states_1")]; + string value_states_1_pad_type_0 = const()[name = string("value_states_1_pad_type_0"), val = string("valid")]; + tensor value_states_1_strides_0 = const()[name = string("value_states_1_strides_0"), val = tensor([1, 1])]; + tensor value_states_1_pad_0 = const()[name = string("value_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_1_dilations_0 = const()[name = string("value_states_1_dilations_0"), val = tensor([1, 1])]; + int32 value_states_1_groups_0 = const()[name = string("value_states_1_groups_0"), val = int32(1)]; + tensor value_states_1 = conv(dilations = value_states_1_dilations_0, groups = value_states_1_groups_0, pad = value_states_1_pad_0, pad_type = value_states_1_pad_type_0, strides = value_states_1_strides_0, weight = model_model_layers_0_self_attn_v_proj_weight_palettized, x = var_830_cast_fp16)[name = string("value_states_1")]; + tensor var_872 = const()[name = string("op_872"), val = tensor([1, 16, 128, 128])]; + tensor var_873 = reshape(shape = var_872, x = query_states_1)[name = string("op_873")]; + tensor var_878 = const()[name = string("op_878"), val = tensor([0, 1, 3, 2])]; + tensor var_883 = const()[name = string("op_883"), val = tensor([1, 8, 128, 128])]; + tensor var_884 = reshape(shape = var_883, x = key_states_1)[name = string("op_884")]; + tensor var_889 = const()[name = string("op_889"), val = tensor([0, 1, 3, 2])]; + tensor var_894 = const()[name = string("op_894"), val = tensor([1, 8, 128, 128])]; + tensor var_895 = reshape(shape = var_894, x = value_states_1)[name = string("op_895")]; + tensor var_900 = const()[name = string("op_900"), val = tensor([0, 1, 3, 2])]; + int32 var_911 = const()[name = string("op_911"), val = int32(-1)]; + fp16 const_6_promoted = const()[name = string("const_6_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_5 = transpose(perm = var_878, x = var_873)[name = string("transpose_126")]; + tensor var_913 = mul(x = hidden_states_5, y = const_6_promoted)[name = string("op_913")]; + bool input_5_interleave_0 = const()[name = string("input_5_interleave_0"), val = bool(false)]; + tensor input_5 = concat(axis = var_911, interleave = input_5_interleave_0, values = (hidden_states_5, var_913))[name = string("input_5")]; + tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; + fp16 var_908_to_fp16 = const()[name = string("op_908_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_908_to_fp16, x = input_5)[name = string("normed_5_cast_fp16")]; + tensor normed_7_begin_0 = const()[name = string("normed_7_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_7_end_0 = const()[name = string("normed_7_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_7_end_mask_0 = const()[name = string("normed_7_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_7 = slice_by_index(begin = normed_7_begin_0, end = normed_7_end_0, end_mask = normed_7_end_mask_0, x = normed_5_cast_fp16)[name = string("normed_7")]; + tensor const_9 = const()[name = string("const_9"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(663501568)))]; + tensor q_1 = mul(x = normed_7, y = const_9)[name = string("q_1")]; + int32 var_936 = const()[name = string("op_936"), val = int32(-1)]; + fp16 const_10_promoted = const()[name = string("const_10_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_7 = transpose(perm = var_889, x = var_884)[name = string("transpose_125")]; + tensor var_938 = mul(x = hidden_states_7, y = const_10_promoted)[name = string("op_938")]; + bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; + tensor input_7 = concat(axis = var_936, interleave = input_7_interleave_0, values = (hidden_states_7, var_938))[name = string("input_7")]; + tensor normed_9_axes_0 = const()[name = string("normed_9_axes_0"), val = tensor([-1])]; + fp16 var_933_to_fp16 = const()[name = string("op_933_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_9_cast_fp16 = layer_norm(axes = normed_9_axes_0, epsilon = var_933_to_fp16, x = input_7)[name = string("normed_9_cast_fp16")]; + tensor normed_11_begin_0 = const()[name = string("normed_11_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_11_end_0 = const()[name = string("normed_11_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_11_end_mask_0 = const()[name = string("normed_11_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_11 = slice_by_index(begin = normed_11_begin_0, end = normed_11_end_0, end_mask = normed_11_end_mask_0, x = normed_9_cast_fp16)[name = string("normed_11")]; + tensor const_13 = const()[name = string("const_13"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(663501888)))]; + tensor k_1 = mul(x = normed_11, y = const_13)[name = string("k_1")]; + tensor var_956 = const()[name = string("op_956"), val = tensor([0, 2, 1, 3])]; + tensor var_962 = const()[name = string("op_962"), val = tensor([0, 2, 1, 3])]; + tensor cos_5 = transpose(perm = var_956, x = cos_1_cast_fp16)[name = string("transpose_124")]; + tensor var_964 = mul(x = q_1, y = cos_5)[name = string("op_964")]; + tensor x1_1_begin_0 = const()[name = string("x1_1_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_1_end_0 = const()[name = string("x1_1_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_1_end_mask_0 = const()[name = string("x1_1_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_1 = slice_by_index(begin = x1_1_begin_0, end = x1_1_end_0, end_mask = x1_1_end_mask_0, x = q_1)[name = string("x1_1")]; + tensor x2_1_begin_0 = const()[name = string("x2_1_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_1_end_0 = const()[name = string("x2_1_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_1_end_mask_0 = const()[name = string("x2_1_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_1 = slice_by_index(begin = x2_1_begin_0, end = x2_1_end_0, end_mask = x2_1_end_mask_0, x = q_1)[name = string("x2_1")]; + fp16 const_16_promoted = const()[name = string("const_16_promoted"), val = fp16(-0x1p+0)]; + tensor var_985 = mul(x = x2_1, y = const_16_promoted)[name = string("op_985")]; + int32 var_987 = const()[name = string("op_987"), val = int32(-1)]; + bool var_988_interleave_0 = const()[name = string("op_988_interleave_0"), val = bool(false)]; + tensor var_988 = concat(axis = var_987, interleave = var_988_interleave_0, values = (var_985, x1_1))[name = string("op_988")]; + tensor sin_5 = transpose(perm = var_962, x = sin_1_cast_fp16)[name = string("transpose_123")]; + tensor var_989 = mul(x = var_988, y = sin_5)[name = string("op_989")]; + tensor query_states_3 = add(x = var_964, y = var_989)[name = string("query_states_3")]; + tensor var_992 = mul(x = k_1, y = cos_5)[name = string("op_992")]; + tensor x1_3_begin_0 = const()[name = string("x1_3_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_3_end_0 = const()[name = string("x1_3_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_3_end_mask_0 = const()[name = string("x1_3_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_3 = slice_by_index(begin = x1_3_begin_0, end = x1_3_end_0, end_mask = x1_3_end_mask_0, x = k_1)[name = string("x1_3")]; + tensor x2_3_begin_0 = const()[name = string("x2_3_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_3_end_0 = const()[name = string("x2_3_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_3_end_mask_0 = const()[name = string("x2_3_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_3 = slice_by_index(begin = x2_3_begin_0, end = x2_3_end_0, end_mask = x2_3_end_mask_0, x = k_1)[name = string("x2_3")]; + fp16 const_19_promoted = const()[name = string("const_19_promoted"), val = fp16(-0x1p+0)]; + tensor var_1013 = mul(x = x2_3, y = const_19_promoted)[name = string("op_1013")]; + int32 var_1015 = const()[name = string("op_1015"), val = int32(-1)]; + bool var_1016_interleave_0 = const()[name = string("op_1016_interleave_0"), val = bool(false)]; + tensor var_1016 = concat(axis = var_1015, interleave = var_1016_interleave_0, values = (var_1013, x1_3))[name = string("op_1016")]; + tensor var_1017 = mul(x = var_1016, y = sin_5)[name = string("op_1017")]; + tensor key_states_3 = add(x = var_992, y = var_1017)[name = string("key_states_3")]; + tensor seq_length_1 = const()[name = string("seq_length_1"), val = tensor([128])]; + tensor var_1039 = add(x = current_pos, y = seq_length_1)[name = string("op_1039")]; + tensor read_state_0 = read_state(input = model_model_kv_cache_0)[name = string("read_state_0")]; + tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([0])]; + tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; + tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; + tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([1])]; + int32 concat_2_axis_0 = const()[name = string("concat_2_axis_0"), val = int32(0)]; + bool concat_2_interleave_0 = const()[name = string("concat_2_interleave_0"), val = bool(false)]; + tensor concat_2 = concat(axis = concat_2_axis_0, interleave = concat_2_interleave_0, values = (expand_dims_0, expand_dims_1, current_pos, expand_dims_3))[name = string("concat_2")]; + tensor concat_3_values1_0 = const()[name = string("concat_3_values1_0"), val = tensor([0])]; + tensor concat_3_values3_0 = const()[name = string("concat_3_values3_0"), val = tensor([0])]; + int32 concat_3_axis_0 = const()[name = string("concat_3_axis_0"), val = int32(0)]; + bool concat_3_interleave_0 = const()[name = string("concat_3_interleave_0"), val = bool(false)]; + tensor concat_3 = concat(axis = concat_3_axis_0, interleave = concat_3_interleave_0, values = (expand_dims_4, concat_3_values1_0, var_1039, concat_3_values3_0))[name = string("concat_3")]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_2, begin_mask = model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0, end = concat_3, end_mask = model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_1_stride_0, update = key_states_3, x = read_state_0)[name = string("model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_84_write_state")]; + tensor coreml_update_state_28 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_84")]; + tensor expand_dims_6 = const()[name = string("expand_dims_6"), val = tensor([28])]; + tensor expand_dims_7 = const()[name = string("expand_dims_7"), val = tensor([0])]; + tensor expand_dims_9 = const()[name = string("expand_dims_9"), val = tensor([0])]; + tensor expand_dims_10 = const()[name = string("expand_dims_10"), val = tensor([29])]; + int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; + bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; + tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_6, expand_dims_7, current_pos, expand_dims_9))[name = string("concat_6")]; + tensor concat_7_values1_0 = const()[name = string("concat_7_values1_0"), val = tensor([0])]; + tensor concat_7_values3_0 = const()[name = string("concat_7_values3_0"), val = tensor([0])]; + int32 concat_7_axis_0 = const()[name = string("concat_7_axis_0"), val = int32(0)]; + bool concat_7_interleave_0 = const()[name = string("concat_7_interleave_0"), val = bool(false)]; + tensor concat_7 = concat(axis = concat_7_axis_0, interleave = concat_7_interleave_0, values = (expand_dims_10, concat_7_values1_0, var_1039, concat_7_values3_0))[name = string("concat_7")]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_3 = transpose(perm = var_900, x = var_895)[name = string("transpose_122")]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_6, begin_mask = model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0, end = concat_7, end_mask = model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_2_stride_0, update = value_states_3, x = coreml_update_state_28)[name = string("model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_85_write_state")]; + tensor coreml_update_state_29 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_85")]; + tensor var_1088_begin_0 = const()[name = string("op_1088_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1088_end_0 = const()[name = string("op_1088_end_0"), val = tensor([1, 8, 1024, 128])]; + tensor var_1088_end_mask_0 = const()[name = string("op_1088_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1088_cast_fp16 = slice_by_index(begin = var_1088_begin_0, end = var_1088_end_0, end_mask = var_1088_end_mask_0, x = coreml_update_state_29)[name = string("op_1088_cast_fp16")]; + tensor K_layer_cache_1_axes_0 = const()[name = string("K_layer_cache_1_axes_0"), val = tensor([0])]; + tensor K_layer_cache_1_cast_fp16 = squeeze(axes = K_layer_cache_1_axes_0, x = var_1088_cast_fp16)[name = string("K_layer_cache_1_cast_fp16")]; + tensor var_1095_begin_0 = const()[name = string("op_1095_begin_0"), val = tensor([28, 0, 0, 0])]; + tensor var_1095_end_0 = const()[name = string("op_1095_end_0"), val = tensor([29, 8, 1024, 128])]; + tensor var_1095_end_mask_0 = const()[name = string("op_1095_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1095_cast_fp16 = slice_by_index(begin = var_1095_begin_0, end = var_1095_end_0, end_mask = var_1095_end_mask_0, x = coreml_update_state_29)[name = string("op_1095_cast_fp16")]; + tensor V_layer_cache_1_axes_0 = const()[name = string("V_layer_cache_1_axes_0"), val = tensor([0])]; + tensor V_layer_cache_1_cast_fp16 = squeeze(axes = V_layer_cache_1_axes_0, x = var_1095_cast_fp16)[name = string("V_layer_cache_1_cast_fp16")]; + tensor x_3_axes_0 = const()[name = string("x_3_axes_0"), val = tensor([1])]; + tensor x_3_cast_fp16 = expand_dims(axes = x_3_axes_0, x = K_layer_cache_1_cast_fp16)[name = string("x_3_cast_fp16")]; + tensor var_1124 = const()[name = string("op_1124"), val = tensor([1, 2, 1, 1])]; + tensor x_5_cast_fp16 = tile(reps = var_1124, x = x_3_cast_fp16)[name = string("x_5_cast_fp16")]; + tensor var_1136 = const()[name = string("op_1136"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_7_cast_fp16 = reshape(shape = var_1136, x = x_5_cast_fp16)[name = string("key_states_7_cast_fp16")]; + tensor x_9_axes_0 = const()[name = string("x_9_axes_0"), val = tensor([1])]; + tensor x_9_cast_fp16 = expand_dims(axes = x_9_axes_0, x = V_layer_cache_1_cast_fp16)[name = string("x_9_cast_fp16")]; + tensor var_1144 = const()[name = string("op_1144"), val = tensor([1, 2, 1, 1])]; + tensor x_11_cast_fp16 = tile(reps = var_1144, x = x_9_cast_fp16)[name = string("x_11_cast_fp16")]; + bool var_1171_transpose_x_0 = const()[name = string("op_1171_transpose_x_0"), val = bool(false)]; + bool var_1171_transpose_y_0 = const()[name = string("op_1171_transpose_y_0"), val = bool(true)]; + tensor var_1171 = matmul(transpose_x = var_1171_transpose_x_0, transpose_y = var_1171_transpose_y_0, x = query_states_3, y = key_states_7_cast_fp16)[name = string("op_1171")]; + fp16 var_1172_to_fp16 = const()[name = string("op_1172_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_1_cast_fp16 = mul(x = var_1171, y = var_1172_to_fp16)[name = string("attn_weights_1_cast_fp16")]; + tensor attn_weights_3_cast_fp16 = add(x = attn_weights_1_cast_fp16, y = causal_mask)[name = string("attn_weights_3_cast_fp16")]; + int32 var_1207 = const()[name = string("op_1207"), val = int32(-1)]; + tensor var_1209_cast_fp16 = softmax(axis = var_1207, x = attn_weights_3_cast_fp16)[name = string("op_1209_cast_fp16")]; + tensor concat_12 = const()[name = string("concat_12"), val = tensor([16, 128, 1024])]; + tensor reshape_0_cast_fp16 = reshape(shape = concat_12, x = var_1209_cast_fp16)[name = string("reshape_0_cast_fp16")]; + tensor concat_13 = const()[name = string("concat_13"), val = tensor([16, 1024, 128])]; + tensor reshape_1_cast_fp16 = reshape(shape = concat_13, x = x_11_cast_fp16)[name = string("reshape_1_cast_fp16")]; + bool matmul_0_transpose_x_0 = const()[name = string("matmul_0_transpose_x_0"), val = bool(false)]; + bool matmul_0_transpose_y_0 = const()[name = string("matmul_0_transpose_y_0"), val = bool(false)]; + tensor matmul_0_cast_fp16 = matmul(transpose_x = matmul_0_transpose_x_0, transpose_y = matmul_0_transpose_y_0, x = reshape_0_cast_fp16, y = reshape_1_cast_fp16)[name = string("matmul_0_cast_fp16")]; + tensor concat_17 = const()[name = string("concat_17"), val = tensor([1, 16, 128, 128])]; + tensor reshape_2_cast_fp16 = reshape(shape = concat_17, x = matmul_0_cast_fp16)[name = string("reshape_2_cast_fp16")]; + tensor var_1221_perm_0 = const()[name = string("op_1221_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1240 = const()[name = string("op_1240"), val = tensor([1, 128, 2048])]; + tensor var_1221_cast_fp16 = transpose(perm = var_1221_perm_0, x = reshape_2_cast_fp16)[name = string("transpose_121")]; + tensor attn_output_5_cast_fp16 = reshape(shape = var_1240, x = var_1221_cast_fp16)[name = string("attn_output_5_cast_fp16")]; + tensor var_1245 = const()[name = string("op_1245"), val = tensor([0, 2, 1])]; + string var_1261_pad_type_0 = const()[name = string("op_1261_pad_type_0"), val = string("valid")]; + int32 var_1261_groups_0 = const()[name = string("op_1261_groups_0"), val = int32(1)]; + tensor var_1261_strides_0 = const()[name = string("op_1261_strides_0"), val = tensor([1])]; + tensor var_1261_pad_0 = const()[name = string("op_1261_pad_0"), val = tensor([0, 0])]; + tensor var_1261_dilations_0 = const()[name = string("op_1261_dilations_0"), val = tensor([1])]; + tensor squeeze_0_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(663502208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667696576))))[name = string("squeeze_0_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_1246_cast_fp16 = transpose(perm = var_1245, x = attn_output_5_cast_fp16)[name = string("transpose_120")]; + tensor var_1261_cast_fp16 = conv(dilations = var_1261_dilations_0, groups = var_1261_groups_0, pad = var_1261_pad_0, pad_type = var_1261_pad_type_0, strides = var_1261_strides_0, weight = squeeze_0_cast_fp16_to_fp32_to_fp16_palettized, x = var_1246_cast_fp16)[name = string("op_1261_cast_fp16")]; + tensor var_1265 = const()[name = string("op_1265"), val = tensor([0, 2, 1])]; + tensor attn_output_9_cast_fp16 = transpose(perm = var_1265, x = var_1261_cast_fp16)[name = string("transpose_119")]; + tensor hidden_states_9_cast_fp16 = add(x = hidden_states, y = attn_output_9_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; + int32 var_1278 = const()[name = string("op_1278"), val = int32(-1)]; + fp16 const_31_promoted_to_fp16 = const()[name = string("const_31_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1280_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_31_promoted_to_fp16)[name = string("op_1280_cast_fp16")]; + bool input_11_interleave_0 = const()[name = string("input_11_interleave_0"), val = bool(false)]; + tensor input_11_cast_fp16 = concat(axis = var_1278, interleave = input_11_interleave_0, values = (hidden_states_9_cast_fp16, var_1280_cast_fp16))[name = string("input_11_cast_fp16")]; + tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; + fp16 var_1275_to_fp16 = const()[name = string("op_1275_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_1275_to_fp16, x = input_11_cast_fp16)[name = string("normed_13_cast_fp16")]; + tensor normed_15_begin_0 = const()[name = string("normed_15_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_15_end_0 = const()[name = string("normed_15_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_15_end_mask_0 = const()[name = string("normed_15_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_15_cast_fp16 = slice_by_index(begin = normed_15_begin_0, end = normed_15_end_0, end_mask = normed_15_end_mask_0, x = normed_13_cast_fp16)[name = string("normed_15_cast_fp16")]; + tensor const_34_promoted_to_fp16 = const()[name = string("const_34_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667827712)))]; + tensor x_13_cast_fp16 = mul(x = normed_15_cast_fp16, y = const_34_promoted_to_fp16)[name = string("x_13_cast_fp16")]; + tensor var_1305 = const()[name = string("op_1305"), val = tensor([0, 2, 1])]; + tensor input_13_axes_0 = const()[name = string("input_13_axes_0"), val = tensor([2])]; + tensor var_1306 = transpose(perm = var_1305, x = x_13_cast_fp16)[name = string("transpose_118")]; + tensor input_13 = expand_dims(axes = input_13_axes_0, x = var_1306)[name = string("input_13")]; + string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; + tensor input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor([1, 1])]; + tensor input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor([1, 1])]; + int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; + tensor input_15 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = model_model_layers_0_mlp_gate_proj_weight_palettized, x = input_13)[name = string("input_15")]; + string b_1_pad_type_0 = const()[name = string("b_1_pad_type_0"), val = string("valid")]; + tensor b_1_strides_0 = const()[name = string("b_1_strides_0"), val = tensor([1, 1])]; + tensor b_1_pad_0 = const()[name = string("b_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_1_dilations_0 = const()[name = string("b_1_dilations_0"), val = tensor([1, 1])]; + int32 b_1_groups_0 = const()[name = string("b_1_groups_0"), val = int32(1)]; + tensor b_1 = conv(dilations = b_1_dilations_0, groups = b_1_groups_0, pad = b_1_pad_0, pad_type = b_1_pad_type_0, strides = b_1_strides_0, weight = model_model_layers_0_mlp_up_proj_weight_palettized, x = input_13)[name = string("b_1")]; + tensor c_1 = silu(x = input_15)[name = string("c_1")]; + tensor input_17 = mul(x = c_1, y = b_1)[name = string("input_17")]; + string e_1_pad_type_0 = const()[name = string("e_1_pad_type_0"), val = string("valid")]; + tensor e_1_strides_0 = const()[name = string("e_1_strides_0"), val = tensor([1, 1])]; + tensor e_1_pad_0 = const()[name = string("e_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_1_dilations_0 = const()[name = string("e_1_dilations_0"), val = tensor([1, 1])]; + int32 e_1_groups_0 = const()[name = string("e_1_groups_0"), val = int32(1)]; + tensor e_1 = conv(dilations = e_1_dilations_0, groups = e_1_groups_0, pad = e_1_pad_0, pad_type = e_1_pad_type_0, strides = e_1_strides_0, weight = model_model_layers_0_mlp_down_proj_weight_palettized, x = input_17)[name = string("e_1")]; + tensor var_1328_axes_0 = const()[name = string("op_1328_axes_0"), val = tensor([2])]; + tensor var_1328 = squeeze(axes = var_1328_axes_0, x = e_1)[name = string("op_1328")]; + tensor var_1329 = const()[name = string("op_1329"), val = tensor([0, 2, 1])]; + tensor var_1330 = transpose(perm = var_1329, x = var_1328)[name = string("transpose_117")]; + tensor hidden_states_11_cast_fp16 = add(x = hidden_states_9_cast_fp16, y = var_1330)[name = string("hidden_states_11_cast_fp16")]; + int32 var_1342 = const()[name = string("op_1342"), val = int32(-1)]; + fp16 const_35_promoted_to_fp16 = const()[name = string("const_35_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1344_cast_fp16 = mul(x = hidden_states_11_cast_fp16, y = const_35_promoted_to_fp16)[name = string("op_1344_cast_fp16")]; + bool input_19_interleave_0 = const()[name = string("input_19_interleave_0"), val = bool(false)]; + tensor input_19_cast_fp16 = concat(axis = var_1342, interleave = input_19_interleave_0, values = (hidden_states_11_cast_fp16, var_1344_cast_fp16))[name = string("input_19_cast_fp16")]; + tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; + fp16 var_1339_to_fp16 = const()[name = string("op_1339_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_1339_to_fp16, x = input_19_cast_fp16)[name = string("normed_17_cast_fp16")]; + tensor normed_19_begin_0 = const()[name = string("normed_19_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_19_end_0 = const()[name = string("normed_19_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_19_end_mask_0 = const()[name = string("normed_19_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_19_cast_fp16 = slice_by_index(begin = normed_19_begin_0, end = normed_19_end_0, end_mask = normed_19_end_mask_0, x = normed_17_cast_fp16)[name = string("normed_19_cast_fp16")]; + tensor const_38_promoted_to_fp16 = const()[name = string("const_38_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667831872)))]; + tensor hidden_states_13_cast_fp16 = mul(x = normed_19_cast_fp16, y = const_38_promoted_to_fp16)[name = string("hidden_states_13_cast_fp16")]; + tensor var_1367 = const()[name = string("op_1367"), val = tensor([0, 2, 1])]; + tensor var_1370_axes_0 = const()[name = string("op_1370_axes_0"), val = tensor([2])]; + tensor var_1368_cast_fp16 = transpose(perm = var_1367, x = hidden_states_13_cast_fp16)[name = string("transpose_116")]; + tensor var_1370_cast_fp16 = expand_dims(axes = var_1370_axes_0, x = var_1368_cast_fp16)[name = string("op_1370_cast_fp16")]; + string query_states_9_pad_type_0 = const()[name = string("query_states_9_pad_type_0"), val = string("valid")]; + tensor query_states_9_strides_0 = const()[name = string("query_states_9_strides_0"), val = tensor([1, 1])]; + tensor query_states_9_pad_0 = const()[name = string("query_states_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_9_dilations_0 = const()[name = string("query_states_9_dilations_0"), val = tensor([1, 1])]; + int32 query_states_9_groups_0 = const()[name = string("query_states_9_groups_0"), val = int32(1)]; + tensor query_states_9 = conv(dilations = query_states_9_dilations_0, groups = query_states_9_groups_0, pad = query_states_9_pad_0, pad_type = query_states_9_pad_type_0, strides = query_states_9_strides_0, weight = model_model_layers_1_self_attn_q_proj_weight_palettized, x = var_1370_cast_fp16)[name = string("query_states_9")]; + string key_states_11_pad_type_0 = const()[name = string("key_states_11_pad_type_0"), val = string("valid")]; + tensor key_states_11_strides_0 = const()[name = string("key_states_11_strides_0"), val = tensor([1, 1])]; + tensor key_states_11_pad_0 = const()[name = string("key_states_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_11_dilations_0 = const()[name = string("key_states_11_dilations_0"), val = tensor([1, 1])]; + int32 key_states_11_groups_0 = const()[name = string("key_states_11_groups_0"), val = int32(1)]; + tensor key_states_11 = conv(dilations = key_states_11_dilations_0, groups = key_states_11_groups_0, pad = key_states_11_pad_0, pad_type = key_states_11_pad_type_0, strides = key_states_11_strides_0, weight = model_model_layers_1_self_attn_k_proj_weight_palettized, x = var_1370_cast_fp16)[name = string("key_states_11")]; + string value_states_9_pad_type_0 = const()[name = string("value_states_9_pad_type_0"), val = string("valid")]; + tensor value_states_9_strides_0 = const()[name = string("value_states_9_strides_0"), val = tensor([1, 1])]; + tensor value_states_9_pad_0 = const()[name = string("value_states_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_9_dilations_0 = const()[name = string("value_states_9_dilations_0"), val = tensor([1, 1])]; + int32 value_states_9_groups_0 = const()[name = string("value_states_9_groups_0"), val = int32(1)]; + tensor value_states_9 = conv(dilations = value_states_9_dilations_0, groups = value_states_9_groups_0, pad = value_states_9_pad_0, pad_type = value_states_9_pad_type_0, strides = value_states_9_strides_0, weight = model_model_layers_1_self_attn_v_proj_weight_palettized, x = var_1370_cast_fp16)[name = string("value_states_9")]; + tensor var_1412 = const()[name = string("op_1412"), val = tensor([1, 16, 128, 128])]; + tensor var_1413 = reshape(shape = var_1412, x = query_states_9)[name = string("op_1413")]; + tensor var_1418 = const()[name = string("op_1418"), val = tensor([0, 1, 3, 2])]; + tensor var_1423 = const()[name = string("op_1423"), val = tensor([1, 8, 128, 128])]; + tensor var_1424 = reshape(shape = var_1423, x = key_states_11)[name = string("op_1424")]; + tensor var_1429 = const()[name = string("op_1429"), val = tensor([0, 1, 3, 2])]; + tensor var_1434 = const()[name = string("op_1434"), val = tensor([1, 8, 128, 128])]; + tensor var_1435 = reshape(shape = var_1434, x = value_states_9)[name = string("op_1435")]; + tensor var_1440 = const()[name = string("op_1440"), val = tensor([0, 1, 3, 2])]; + int32 var_1451 = const()[name = string("op_1451"), val = int32(-1)]; + fp16 const_40_promoted = const()[name = string("const_40_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_15 = transpose(perm = var_1418, x = var_1413)[name = string("transpose_115")]; + tensor var_1453 = mul(x = hidden_states_15, y = const_40_promoted)[name = string("op_1453")]; + bool input_23_interleave_0 = const()[name = string("input_23_interleave_0"), val = bool(false)]; + tensor input_23 = concat(axis = var_1451, interleave = input_23_interleave_0, values = (hidden_states_15, var_1453))[name = string("input_23")]; + tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; + fp16 var_1448_to_fp16 = const()[name = string("op_1448_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_1448_to_fp16, x = input_23)[name = string("normed_21_cast_fp16")]; + tensor normed_23_begin_0 = const()[name = string("normed_23_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_23_end_0 = const()[name = string("normed_23_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_23_end_mask_0 = const()[name = string("normed_23_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_23 = slice_by_index(begin = normed_23_begin_0, end = normed_23_end_0, end_mask = normed_23_end_mask_0, x = normed_21_cast_fp16)[name = string("normed_23")]; + tensor const_43 = const()[name = string("const_43"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667836032)))]; + tensor q_3 = mul(x = normed_23, y = const_43)[name = string("q_3")]; + int32 var_1476 = const()[name = string("op_1476"), val = int32(-1)]; + fp16 const_44_promoted = const()[name = string("const_44_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_17 = transpose(perm = var_1429, x = var_1424)[name = string("transpose_114")]; + tensor var_1478 = mul(x = hidden_states_17, y = const_44_promoted)[name = string("op_1478")]; + bool input_25_interleave_0 = const()[name = string("input_25_interleave_0"), val = bool(false)]; + tensor input_25 = concat(axis = var_1476, interleave = input_25_interleave_0, values = (hidden_states_17, var_1478))[name = string("input_25")]; + tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; + fp16 var_1473_to_fp16 = const()[name = string("op_1473_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_1473_to_fp16, x = input_25)[name = string("normed_25_cast_fp16")]; + tensor normed_27_begin_0 = const()[name = string("normed_27_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_27_end_0 = const()[name = string("normed_27_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_27_end_mask_0 = const()[name = string("normed_27_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_27 = slice_by_index(begin = normed_27_begin_0, end = normed_27_end_0, end_mask = normed_27_end_mask_0, x = normed_25_cast_fp16)[name = string("normed_27")]; + tensor const_47 = const()[name = string("const_47"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667836352)))]; + tensor k_3 = mul(x = normed_27, y = const_47)[name = string("k_3")]; + tensor var_1504 = mul(x = q_3, y = cos_5)[name = string("op_1504")]; + tensor x1_5_begin_0 = const()[name = string("x1_5_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_5_end_0 = const()[name = string("x1_5_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_5_end_mask_0 = const()[name = string("x1_5_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_5 = slice_by_index(begin = x1_5_begin_0, end = x1_5_end_0, end_mask = x1_5_end_mask_0, x = q_3)[name = string("x1_5")]; + tensor x2_5_begin_0 = const()[name = string("x2_5_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_5_end_0 = const()[name = string("x2_5_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_5_end_mask_0 = const()[name = string("x2_5_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_5 = slice_by_index(begin = x2_5_begin_0, end = x2_5_end_0, end_mask = x2_5_end_mask_0, x = q_3)[name = string("x2_5")]; + fp16 const_50_promoted = const()[name = string("const_50_promoted"), val = fp16(-0x1p+0)]; + tensor var_1525 = mul(x = x2_5, y = const_50_promoted)[name = string("op_1525")]; + int32 var_1527 = const()[name = string("op_1527"), val = int32(-1)]; + bool var_1528_interleave_0 = const()[name = string("op_1528_interleave_0"), val = bool(false)]; + tensor var_1528 = concat(axis = var_1527, interleave = var_1528_interleave_0, values = (var_1525, x1_5))[name = string("op_1528")]; + tensor var_1529 = mul(x = var_1528, y = sin_5)[name = string("op_1529")]; + tensor query_states_11 = add(x = var_1504, y = var_1529)[name = string("query_states_11")]; + tensor var_1532 = mul(x = k_3, y = cos_5)[name = string("op_1532")]; + tensor x1_7_begin_0 = const()[name = string("x1_7_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_7_end_0 = const()[name = string("x1_7_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_7_end_mask_0 = const()[name = string("x1_7_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_7 = slice_by_index(begin = x1_7_begin_0, end = x1_7_end_0, end_mask = x1_7_end_mask_0, x = k_3)[name = string("x1_7")]; + tensor x2_7_begin_0 = const()[name = string("x2_7_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_7_end_0 = const()[name = string("x2_7_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_7_end_mask_0 = const()[name = string("x2_7_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_7 = slice_by_index(begin = x2_7_begin_0, end = x2_7_end_0, end_mask = x2_7_end_mask_0, x = k_3)[name = string("x2_7")]; + fp16 const_53_promoted = const()[name = string("const_53_promoted"), val = fp16(-0x1p+0)]; + tensor var_1553 = mul(x = x2_7, y = const_53_promoted)[name = string("op_1553")]; + int32 var_1555 = const()[name = string("op_1555"), val = int32(-1)]; + bool var_1556_interleave_0 = const()[name = string("op_1556_interleave_0"), val = bool(false)]; + tensor var_1556 = concat(axis = var_1555, interleave = var_1556_interleave_0, values = (var_1553, x1_7))[name = string("op_1556")]; + tensor var_1557 = mul(x = var_1556, y = sin_5)[name = string("op_1557")]; + tensor key_states_13 = add(x = var_1532, y = var_1557)[name = string("key_states_13")]; + tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([1])]; + tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; + tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; + tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([2])]; + int32 concat_20_axis_0 = const()[name = string("concat_20_axis_0"), val = int32(0)]; + bool concat_20_interleave_0 = const()[name = string("concat_20_interleave_0"), val = bool(false)]; + tensor concat_20 = concat(axis = concat_20_axis_0, interleave = concat_20_interleave_0, values = (expand_dims_12, expand_dims_13, current_pos, expand_dims_15))[name = string("concat_20")]; + tensor concat_21_values1_0 = const()[name = string("concat_21_values1_0"), val = tensor([0])]; + tensor concat_21_values3_0 = const()[name = string("concat_21_values3_0"), val = tensor([0])]; + int32 concat_21_axis_0 = const()[name = string("concat_21_axis_0"), val = int32(0)]; + bool concat_21_interleave_0 = const()[name = string("concat_21_interleave_0"), val = bool(false)]; + tensor concat_21 = concat(axis = concat_21_axis_0, interleave = concat_21_interleave_0, values = (expand_dims_16, concat_21_values1_0, var_1039, concat_21_values3_0))[name = string("concat_21")]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_20, begin_mask = model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0, end = concat_21, end_mask = model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_3_stride_0, update = key_states_13, x = coreml_update_state_29)[name = string("model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_86_write_state")]; + tensor coreml_update_state_30 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_86")]; + tensor expand_dims_18 = const()[name = string("expand_dims_18"), val = tensor([29])]; + tensor expand_dims_19 = const()[name = string("expand_dims_19"), val = tensor([0])]; + tensor expand_dims_21 = const()[name = string("expand_dims_21"), val = tensor([0])]; + tensor expand_dims_22 = const()[name = string("expand_dims_22"), val = tensor([30])]; + int32 concat_24_axis_0 = const()[name = string("concat_24_axis_0"), val = int32(0)]; + bool concat_24_interleave_0 = const()[name = string("concat_24_interleave_0"), val = bool(false)]; + tensor concat_24 = concat(axis = concat_24_axis_0, interleave = concat_24_interleave_0, values = (expand_dims_18, expand_dims_19, current_pos, expand_dims_21))[name = string("concat_24")]; + tensor concat_25_values1_0 = const()[name = string("concat_25_values1_0"), val = tensor([0])]; + tensor concat_25_values3_0 = const()[name = string("concat_25_values3_0"), val = tensor([0])]; + int32 concat_25_axis_0 = const()[name = string("concat_25_axis_0"), val = int32(0)]; + bool concat_25_interleave_0 = const()[name = string("concat_25_interleave_0"), val = bool(false)]; + tensor concat_25 = concat(axis = concat_25_axis_0, interleave = concat_25_interleave_0, values = (expand_dims_22, concat_25_values1_0, var_1039, concat_25_values3_0))[name = string("concat_25")]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_11 = transpose(perm = var_1440, x = var_1435)[name = string("transpose_113")]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_24, begin_mask = model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0, end = concat_25, end_mask = model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_4_stride_0, update = value_states_11, x = coreml_update_state_30)[name = string("model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_87_write_state")]; + tensor coreml_update_state_31 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_87")]; + tensor var_1628_begin_0 = const()[name = string("op_1628_begin_0"), val = tensor([1, 0, 0, 0])]; + tensor var_1628_end_0 = const()[name = string("op_1628_end_0"), val = tensor([2, 8, 1024, 128])]; + tensor var_1628_end_mask_0 = const()[name = string("op_1628_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1628_cast_fp16 = slice_by_index(begin = var_1628_begin_0, end = var_1628_end_0, end_mask = var_1628_end_mask_0, x = coreml_update_state_31)[name = string("op_1628_cast_fp16")]; + tensor K_layer_cache_3_axes_0 = const()[name = string("K_layer_cache_3_axes_0"), val = tensor([0])]; + tensor K_layer_cache_3_cast_fp16 = squeeze(axes = K_layer_cache_3_axes_0, x = var_1628_cast_fp16)[name = string("K_layer_cache_3_cast_fp16")]; + tensor var_1635_begin_0 = const()[name = string("op_1635_begin_0"), val = tensor([29, 0, 0, 0])]; + tensor var_1635_end_0 = const()[name = string("op_1635_end_0"), val = tensor([30, 8, 1024, 128])]; + tensor var_1635_end_mask_0 = const()[name = string("op_1635_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1635_cast_fp16 = slice_by_index(begin = var_1635_begin_0, end = var_1635_end_0, end_mask = var_1635_end_mask_0, x = coreml_update_state_31)[name = string("op_1635_cast_fp16")]; + tensor V_layer_cache_3_axes_0 = const()[name = string("V_layer_cache_3_axes_0"), val = tensor([0])]; + tensor V_layer_cache_3_cast_fp16 = squeeze(axes = V_layer_cache_3_axes_0, x = var_1635_cast_fp16)[name = string("V_layer_cache_3_cast_fp16")]; + tensor x_19_axes_0 = const()[name = string("x_19_axes_0"), val = tensor([1])]; + tensor x_19_cast_fp16 = expand_dims(axes = x_19_axes_0, x = K_layer_cache_3_cast_fp16)[name = string("x_19_cast_fp16")]; + tensor var_1664 = const()[name = string("op_1664"), val = tensor([1, 2, 1, 1])]; + tensor x_21_cast_fp16 = tile(reps = var_1664, x = x_19_cast_fp16)[name = string("x_21_cast_fp16")]; + tensor var_1676 = const()[name = string("op_1676"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_17_cast_fp16 = reshape(shape = var_1676, x = x_21_cast_fp16)[name = string("key_states_17_cast_fp16")]; + tensor x_25_axes_0 = const()[name = string("x_25_axes_0"), val = tensor([1])]; + tensor x_25_cast_fp16 = expand_dims(axes = x_25_axes_0, x = V_layer_cache_3_cast_fp16)[name = string("x_25_cast_fp16")]; + tensor var_1684 = const()[name = string("op_1684"), val = tensor([1, 2, 1, 1])]; + tensor x_27_cast_fp16 = tile(reps = var_1684, x = x_25_cast_fp16)[name = string("x_27_cast_fp16")]; + bool var_1711_transpose_x_0 = const()[name = string("op_1711_transpose_x_0"), val = bool(false)]; + bool var_1711_transpose_y_0 = const()[name = string("op_1711_transpose_y_0"), val = bool(true)]; + tensor var_1711 = matmul(transpose_x = var_1711_transpose_x_0, transpose_y = var_1711_transpose_y_0, x = query_states_11, y = key_states_17_cast_fp16)[name = string("op_1711")]; + fp16 var_1712_to_fp16 = const()[name = string("op_1712_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_5_cast_fp16 = mul(x = var_1711, y = var_1712_to_fp16)[name = string("attn_weights_5_cast_fp16")]; + tensor attn_weights_7_cast_fp16 = add(x = attn_weights_5_cast_fp16, y = causal_mask)[name = string("attn_weights_7_cast_fp16")]; + int32 var_1747 = const()[name = string("op_1747"), val = int32(-1)]; + tensor var_1749_cast_fp16 = softmax(axis = var_1747, x = attn_weights_7_cast_fp16)[name = string("op_1749_cast_fp16")]; + tensor concat_30 = const()[name = string("concat_30"), val = tensor([16, 128, 1024])]; + tensor reshape_3_cast_fp16 = reshape(shape = concat_30, x = var_1749_cast_fp16)[name = string("reshape_3_cast_fp16")]; + tensor concat_31 = const()[name = string("concat_31"), val = tensor([16, 1024, 128])]; + tensor reshape_4_cast_fp16 = reshape(shape = concat_31, x = x_27_cast_fp16)[name = string("reshape_4_cast_fp16")]; + bool matmul_1_transpose_x_0 = const()[name = string("matmul_1_transpose_x_0"), val = bool(false)]; + bool matmul_1_transpose_y_0 = const()[name = string("matmul_1_transpose_y_0"), val = bool(false)]; + tensor matmul_1_cast_fp16 = matmul(transpose_x = matmul_1_transpose_x_0, transpose_y = matmul_1_transpose_y_0, x = reshape_3_cast_fp16, y = reshape_4_cast_fp16)[name = string("matmul_1_cast_fp16")]; + tensor concat_35 = const()[name = string("concat_35"), val = tensor([1, 16, 128, 128])]; + tensor reshape_5_cast_fp16 = reshape(shape = concat_35, x = matmul_1_cast_fp16)[name = string("reshape_5_cast_fp16")]; + tensor var_1761_perm_0 = const()[name = string("op_1761_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1780 = const()[name = string("op_1780"), val = tensor([1, 128, 2048])]; + tensor var_1761_cast_fp16 = transpose(perm = var_1761_perm_0, x = reshape_5_cast_fp16)[name = string("transpose_112")]; + tensor attn_output_15_cast_fp16 = reshape(shape = var_1780, x = var_1761_cast_fp16)[name = string("attn_output_15_cast_fp16")]; + tensor var_1785 = const()[name = string("op_1785"), val = tensor([0, 2, 1])]; + string var_1801_pad_type_0 = const()[name = string("op_1801_pad_type_0"), val = string("valid")]; + int32 var_1801_groups_0 = const()[name = string("op_1801_groups_0"), val = int32(1)]; + tensor var_1801_strides_0 = const()[name = string("op_1801_strides_0"), val = tensor([1])]; + tensor var_1801_pad_0 = const()[name = string("op_1801_pad_0"), val = tensor([0, 0])]; + tensor var_1801_dilations_0 = const()[name = string("op_1801_dilations_0"), val = tensor([1])]; + tensor squeeze_1_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667836672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672031040))))[name = string("squeeze_1_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_1786_cast_fp16 = transpose(perm = var_1785, x = attn_output_15_cast_fp16)[name = string("transpose_111")]; + tensor var_1801_cast_fp16 = conv(dilations = var_1801_dilations_0, groups = var_1801_groups_0, pad = var_1801_pad_0, pad_type = var_1801_pad_type_0, strides = var_1801_strides_0, weight = squeeze_1_cast_fp16_to_fp32_to_fp16_palettized, x = var_1786_cast_fp16)[name = string("op_1801_cast_fp16")]; + tensor var_1805 = const()[name = string("op_1805"), val = tensor([0, 2, 1])]; + tensor attn_output_19_cast_fp16 = transpose(perm = var_1805, x = var_1801_cast_fp16)[name = string("transpose_110")]; + tensor hidden_states_19_cast_fp16 = add(x = hidden_states_11_cast_fp16, y = attn_output_19_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; + int32 var_1818 = const()[name = string("op_1818"), val = int32(-1)]; + fp16 const_65_promoted_to_fp16 = const()[name = string("const_65_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1820_cast_fp16 = mul(x = hidden_states_19_cast_fp16, y = const_65_promoted_to_fp16)[name = string("op_1820_cast_fp16")]; + bool input_29_interleave_0 = const()[name = string("input_29_interleave_0"), val = bool(false)]; + tensor input_29_cast_fp16 = concat(axis = var_1818, interleave = input_29_interleave_0, values = (hidden_states_19_cast_fp16, var_1820_cast_fp16))[name = string("input_29_cast_fp16")]; + tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; + fp16 var_1815_to_fp16 = const()[name = string("op_1815_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_1815_to_fp16, x = input_29_cast_fp16)[name = string("normed_29_cast_fp16")]; + tensor normed_31_begin_0 = const()[name = string("normed_31_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_31_end_0 = const()[name = string("normed_31_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_31_end_mask_0 = const()[name = string("normed_31_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_31_cast_fp16 = slice_by_index(begin = normed_31_begin_0, end = normed_31_end_0, end_mask = normed_31_end_mask_0, x = normed_29_cast_fp16)[name = string("normed_31_cast_fp16")]; + tensor const_68_promoted_to_fp16 = const()[name = string("const_68_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672162176)))]; + tensor x_29_cast_fp16 = mul(x = normed_31_cast_fp16, y = const_68_promoted_to_fp16)[name = string("x_29_cast_fp16")]; + tensor var_1845 = const()[name = string("op_1845"), val = tensor([0, 2, 1])]; + tensor input_31_axes_0 = const()[name = string("input_31_axes_0"), val = tensor([2])]; + tensor var_1846 = transpose(perm = var_1845, x = x_29_cast_fp16)[name = string("transpose_109")]; + tensor input_31 = expand_dims(axes = input_31_axes_0, x = var_1846)[name = string("input_31")]; + string input_33_pad_type_0 = const()[name = string("input_33_pad_type_0"), val = string("valid")]; + tensor input_33_strides_0 = const()[name = string("input_33_strides_0"), val = tensor([1, 1])]; + tensor input_33_pad_0 = const()[name = string("input_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_33_dilations_0 = const()[name = string("input_33_dilations_0"), val = tensor([1, 1])]; + int32 input_33_groups_0 = const()[name = string("input_33_groups_0"), val = int32(1)]; + tensor input_33 = conv(dilations = input_33_dilations_0, groups = input_33_groups_0, pad = input_33_pad_0, pad_type = input_33_pad_type_0, strides = input_33_strides_0, weight = model_model_layers_1_mlp_gate_proj_weight_palettized, x = input_31)[name = string("input_33")]; + string b_3_pad_type_0 = const()[name = string("b_3_pad_type_0"), val = string("valid")]; + tensor b_3_strides_0 = const()[name = string("b_3_strides_0"), val = tensor([1, 1])]; + tensor b_3_pad_0 = const()[name = string("b_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_3_dilations_0 = const()[name = string("b_3_dilations_0"), val = tensor([1, 1])]; + int32 b_3_groups_0 = const()[name = string("b_3_groups_0"), val = int32(1)]; + tensor b_3 = conv(dilations = b_3_dilations_0, groups = b_3_groups_0, pad = b_3_pad_0, pad_type = b_3_pad_type_0, strides = b_3_strides_0, weight = model_model_layers_1_mlp_up_proj_weight_palettized, x = input_31)[name = string("b_3")]; + tensor c_3 = silu(x = input_33)[name = string("c_3")]; + tensor input_35 = mul(x = c_3, y = b_3)[name = string("input_35")]; + string e_3_pad_type_0 = const()[name = string("e_3_pad_type_0"), val = string("valid")]; + tensor e_3_strides_0 = const()[name = string("e_3_strides_0"), val = tensor([1, 1])]; + tensor e_3_pad_0 = const()[name = string("e_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_3_dilations_0 = const()[name = string("e_3_dilations_0"), val = tensor([1, 1])]; + int32 e_3_groups_0 = const()[name = string("e_3_groups_0"), val = int32(1)]; + tensor e_3 = conv(dilations = e_3_dilations_0, groups = e_3_groups_0, pad = e_3_pad_0, pad_type = e_3_pad_type_0, strides = e_3_strides_0, weight = model_model_layers_1_mlp_down_proj_weight_palettized, x = input_35)[name = string("e_3")]; + tensor var_1868_axes_0 = const()[name = string("op_1868_axes_0"), val = tensor([2])]; + tensor var_1868 = squeeze(axes = var_1868_axes_0, x = e_3)[name = string("op_1868")]; + tensor var_1869 = const()[name = string("op_1869"), val = tensor([0, 2, 1])]; + tensor var_1870 = transpose(perm = var_1869, x = var_1868)[name = string("transpose_108")]; + tensor hidden_states_21_cast_fp16 = add(x = hidden_states_19_cast_fp16, y = var_1870)[name = string("hidden_states_21_cast_fp16")]; + int32 var_1882 = const()[name = string("op_1882"), val = int32(-1)]; + fp16 const_69_promoted_to_fp16 = const()[name = string("const_69_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1884_cast_fp16 = mul(x = hidden_states_21_cast_fp16, y = const_69_promoted_to_fp16)[name = string("op_1884_cast_fp16")]; + bool input_37_interleave_0 = const()[name = string("input_37_interleave_0"), val = bool(false)]; + tensor input_37_cast_fp16 = concat(axis = var_1882, interleave = input_37_interleave_0, values = (hidden_states_21_cast_fp16, var_1884_cast_fp16))[name = string("input_37_cast_fp16")]; + tensor normed_33_axes_0 = const()[name = string("normed_33_axes_0"), val = tensor([-1])]; + fp16 var_1879_to_fp16 = const()[name = string("op_1879_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_33_cast_fp16 = layer_norm(axes = normed_33_axes_0, epsilon = var_1879_to_fp16, x = input_37_cast_fp16)[name = string("normed_33_cast_fp16")]; + tensor normed_35_begin_0 = const()[name = string("normed_35_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_35_end_0 = const()[name = string("normed_35_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_35_end_mask_0 = const()[name = string("normed_35_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_35_cast_fp16 = slice_by_index(begin = normed_35_begin_0, end = normed_35_end_0, end_mask = normed_35_end_mask_0, x = normed_33_cast_fp16)[name = string("normed_35_cast_fp16")]; + tensor const_72_promoted_to_fp16 = const()[name = string("const_72_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672166336)))]; + tensor hidden_states_23_cast_fp16 = mul(x = normed_35_cast_fp16, y = const_72_promoted_to_fp16)[name = string("hidden_states_23_cast_fp16")]; + tensor var_1907 = const()[name = string("op_1907"), val = tensor([0, 2, 1])]; + tensor var_1910_axes_0 = const()[name = string("op_1910_axes_0"), val = tensor([2])]; + tensor var_1908_cast_fp16 = transpose(perm = var_1907, x = hidden_states_23_cast_fp16)[name = string("transpose_107")]; + tensor var_1910_cast_fp16 = expand_dims(axes = var_1910_axes_0, x = var_1908_cast_fp16)[name = string("op_1910_cast_fp16")]; + string query_states_17_pad_type_0 = const()[name = string("query_states_17_pad_type_0"), val = string("valid")]; + tensor query_states_17_strides_0 = const()[name = string("query_states_17_strides_0"), val = tensor([1, 1])]; + tensor query_states_17_pad_0 = const()[name = string("query_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_17_dilations_0 = const()[name = string("query_states_17_dilations_0"), val = tensor([1, 1])]; + int32 query_states_17_groups_0 = const()[name = string("query_states_17_groups_0"), val = int32(1)]; + tensor query_states_17 = conv(dilations = query_states_17_dilations_0, groups = query_states_17_groups_0, pad = query_states_17_pad_0, pad_type = query_states_17_pad_type_0, strides = query_states_17_strides_0, weight = model_model_layers_2_self_attn_q_proj_weight_palettized, x = var_1910_cast_fp16)[name = string("query_states_17")]; + string key_states_21_pad_type_0 = const()[name = string("key_states_21_pad_type_0"), val = string("valid")]; + tensor key_states_21_strides_0 = const()[name = string("key_states_21_strides_0"), val = tensor([1, 1])]; + tensor key_states_21_pad_0 = const()[name = string("key_states_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_21_dilations_0 = const()[name = string("key_states_21_dilations_0"), val = tensor([1, 1])]; + int32 key_states_21_groups_0 = const()[name = string("key_states_21_groups_0"), val = int32(1)]; + tensor key_states_21 = conv(dilations = key_states_21_dilations_0, groups = key_states_21_groups_0, pad = key_states_21_pad_0, pad_type = key_states_21_pad_type_0, strides = key_states_21_strides_0, weight = model_model_layers_2_self_attn_k_proj_weight_palettized, x = var_1910_cast_fp16)[name = string("key_states_21")]; + string value_states_17_pad_type_0 = const()[name = string("value_states_17_pad_type_0"), val = string("valid")]; + tensor value_states_17_strides_0 = const()[name = string("value_states_17_strides_0"), val = tensor([1, 1])]; + tensor value_states_17_pad_0 = const()[name = string("value_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_17_dilations_0 = const()[name = string("value_states_17_dilations_0"), val = tensor([1, 1])]; + int32 value_states_17_groups_0 = const()[name = string("value_states_17_groups_0"), val = int32(1)]; + tensor value_states_17 = conv(dilations = value_states_17_dilations_0, groups = value_states_17_groups_0, pad = value_states_17_pad_0, pad_type = value_states_17_pad_type_0, strides = value_states_17_strides_0, weight = model_model_layers_2_self_attn_v_proj_weight_palettized, x = var_1910_cast_fp16)[name = string("value_states_17")]; + tensor var_1952 = const()[name = string("op_1952"), val = tensor([1, 16, 128, 128])]; + tensor var_1953 = reshape(shape = var_1952, x = query_states_17)[name = string("op_1953")]; + tensor var_1958 = const()[name = string("op_1958"), val = tensor([0, 1, 3, 2])]; + tensor var_1963 = const()[name = string("op_1963"), val = tensor([1, 8, 128, 128])]; + tensor var_1964 = reshape(shape = var_1963, x = key_states_21)[name = string("op_1964")]; + tensor var_1969 = const()[name = string("op_1969"), val = tensor([0, 1, 3, 2])]; + tensor var_1974 = const()[name = string("op_1974"), val = tensor([1, 8, 128, 128])]; + tensor var_1975 = reshape(shape = var_1974, x = value_states_17)[name = string("op_1975")]; + tensor var_1980 = const()[name = string("op_1980"), val = tensor([0, 1, 3, 2])]; + int32 var_1991 = const()[name = string("op_1991"), val = int32(-1)]; + fp16 const_74_promoted = const()[name = string("const_74_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_25 = transpose(perm = var_1958, x = var_1953)[name = string("transpose_106")]; + tensor var_1993 = mul(x = hidden_states_25, y = const_74_promoted)[name = string("op_1993")]; + bool input_41_interleave_0 = const()[name = string("input_41_interleave_0"), val = bool(false)]; + tensor input_41 = concat(axis = var_1991, interleave = input_41_interleave_0, values = (hidden_states_25, var_1993))[name = string("input_41")]; + tensor normed_37_axes_0 = const()[name = string("normed_37_axes_0"), val = tensor([-1])]; + fp16 var_1988_to_fp16 = const()[name = string("op_1988_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_37_cast_fp16 = layer_norm(axes = normed_37_axes_0, epsilon = var_1988_to_fp16, x = input_41)[name = string("normed_37_cast_fp16")]; + tensor normed_39_begin_0 = const()[name = string("normed_39_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_39_end_0 = const()[name = string("normed_39_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_39_end_mask_0 = const()[name = string("normed_39_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_39 = slice_by_index(begin = normed_39_begin_0, end = normed_39_end_0, end_mask = normed_39_end_mask_0, x = normed_37_cast_fp16)[name = string("normed_39")]; + tensor const_77 = const()[name = string("const_77"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672170496)))]; + tensor q_5 = mul(x = normed_39, y = const_77)[name = string("q_5")]; + int32 var_2016 = const()[name = string("op_2016"), val = int32(-1)]; + fp16 const_78_promoted = const()[name = string("const_78_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_27 = transpose(perm = var_1969, x = var_1964)[name = string("transpose_105")]; + tensor var_2018 = mul(x = hidden_states_27, y = const_78_promoted)[name = string("op_2018")]; + bool input_43_interleave_0 = const()[name = string("input_43_interleave_0"), val = bool(false)]; + tensor input_43 = concat(axis = var_2016, interleave = input_43_interleave_0, values = (hidden_states_27, var_2018))[name = string("input_43")]; + tensor normed_41_axes_0 = const()[name = string("normed_41_axes_0"), val = tensor([-1])]; + fp16 var_2013_to_fp16 = const()[name = string("op_2013_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_41_cast_fp16 = layer_norm(axes = normed_41_axes_0, epsilon = var_2013_to_fp16, x = input_43)[name = string("normed_41_cast_fp16")]; + tensor normed_43_begin_0 = const()[name = string("normed_43_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_43_end_0 = const()[name = string("normed_43_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_43_end_mask_0 = const()[name = string("normed_43_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_43 = slice_by_index(begin = normed_43_begin_0, end = normed_43_end_0, end_mask = normed_43_end_mask_0, x = normed_41_cast_fp16)[name = string("normed_43")]; + tensor const_81 = const()[name = string("const_81"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672170816)))]; + tensor k_5 = mul(x = normed_43, y = const_81)[name = string("k_5")]; + tensor var_2044 = mul(x = q_5, y = cos_5)[name = string("op_2044")]; + tensor x1_9_begin_0 = const()[name = string("x1_9_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_9_end_0 = const()[name = string("x1_9_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_9_end_mask_0 = const()[name = string("x1_9_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_9 = slice_by_index(begin = x1_9_begin_0, end = x1_9_end_0, end_mask = x1_9_end_mask_0, x = q_5)[name = string("x1_9")]; + tensor x2_9_begin_0 = const()[name = string("x2_9_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_9_end_0 = const()[name = string("x2_9_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_9_end_mask_0 = const()[name = string("x2_9_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_9 = slice_by_index(begin = x2_9_begin_0, end = x2_9_end_0, end_mask = x2_9_end_mask_0, x = q_5)[name = string("x2_9")]; + fp16 const_84_promoted = const()[name = string("const_84_promoted"), val = fp16(-0x1p+0)]; + tensor var_2065 = mul(x = x2_9, y = const_84_promoted)[name = string("op_2065")]; + int32 var_2067 = const()[name = string("op_2067"), val = int32(-1)]; + bool var_2068_interleave_0 = const()[name = string("op_2068_interleave_0"), val = bool(false)]; + tensor var_2068 = concat(axis = var_2067, interleave = var_2068_interleave_0, values = (var_2065, x1_9))[name = string("op_2068")]; + tensor var_2069 = mul(x = var_2068, y = sin_5)[name = string("op_2069")]; + tensor query_states_19 = add(x = var_2044, y = var_2069)[name = string("query_states_19")]; + tensor var_2072 = mul(x = k_5, y = cos_5)[name = string("op_2072")]; + tensor x1_11_begin_0 = const()[name = string("x1_11_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_11_end_0 = const()[name = string("x1_11_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_11_end_mask_0 = const()[name = string("x1_11_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_11 = slice_by_index(begin = x1_11_begin_0, end = x1_11_end_0, end_mask = x1_11_end_mask_0, x = k_5)[name = string("x1_11")]; + tensor x2_11_begin_0 = const()[name = string("x2_11_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_11_end_0 = const()[name = string("x2_11_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_11_end_mask_0 = const()[name = string("x2_11_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_11 = slice_by_index(begin = x2_11_begin_0, end = x2_11_end_0, end_mask = x2_11_end_mask_0, x = k_5)[name = string("x2_11")]; + fp16 const_87_promoted = const()[name = string("const_87_promoted"), val = fp16(-0x1p+0)]; + tensor var_2093 = mul(x = x2_11, y = const_87_promoted)[name = string("op_2093")]; + int32 var_2095 = const()[name = string("op_2095"), val = int32(-1)]; + bool var_2096_interleave_0 = const()[name = string("op_2096_interleave_0"), val = bool(false)]; + tensor var_2096 = concat(axis = var_2095, interleave = var_2096_interleave_0, values = (var_2093, x1_11))[name = string("op_2096")]; + tensor var_2097 = mul(x = var_2096, y = sin_5)[name = string("op_2097")]; + tensor key_states_23 = add(x = var_2072, y = var_2097)[name = string("key_states_23")]; + tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([2])]; + tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; + tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; + tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([3])]; + int32 concat_38_axis_0 = const()[name = string("concat_38_axis_0"), val = int32(0)]; + bool concat_38_interleave_0 = const()[name = string("concat_38_interleave_0"), val = bool(false)]; + tensor concat_38 = concat(axis = concat_38_axis_0, interleave = concat_38_interleave_0, values = (expand_dims_24, expand_dims_25, current_pos, expand_dims_27))[name = string("concat_38")]; + tensor concat_39_values1_0 = const()[name = string("concat_39_values1_0"), val = tensor([0])]; + tensor concat_39_values3_0 = const()[name = string("concat_39_values3_0"), val = tensor([0])]; + int32 concat_39_axis_0 = const()[name = string("concat_39_axis_0"), val = int32(0)]; + bool concat_39_interleave_0 = const()[name = string("concat_39_interleave_0"), val = bool(false)]; + tensor concat_39 = concat(axis = concat_39_axis_0, interleave = concat_39_interleave_0, values = (expand_dims_28, concat_39_values1_0, var_1039, concat_39_values3_0))[name = string("concat_39")]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_38, begin_mask = model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0, end = concat_39, end_mask = model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_5_stride_0, update = key_states_23, x = coreml_update_state_31)[name = string("model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_88_write_state")]; + tensor coreml_update_state_32 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_88")]; + tensor expand_dims_30 = const()[name = string("expand_dims_30"), val = tensor([30])]; + tensor expand_dims_31 = const()[name = string("expand_dims_31"), val = tensor([0])]; + tensor expand_dims_33 = const()[name = string("expand_dims_33"), val = tensor([0])]; + tensor expand_dims_34 = const()[name = string("expand_dims_34"), val = tensor([31])]; + int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; + bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; + tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_30, expand_dims_31, current_pos, expand_dims_33))[name = string("concat_42")]; + tensor concat_43_values1_0 = const()[name = string("concat_43_values1_0"), val = tensor([0])]; + tensor concat_43_values3_0 = const()[name = string("concat_43_values3_0"), val = tensor([0])]; + int32 concat_43_axis_0 = const()[name = string("concat_43_axis_0"), val = int32(0)]; + bool concat_43_interleave_0 = const()[name = string("concat_43_interleave_0"), val = bool(false)]; + tensor concat_43 = concat(axis = concat_43_axis_0, interleave = concat_43_interleave_0, values = (expand_dims_34, concat_43_values1_0, var_1039, concat_43_values3_0))[name = string("concat_43")]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_19 = transpose(perm = var_1980, x = var_1975)[name = string("transpose_104")]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_42, begin_mask = model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0, end = concat_43, end_mask = model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_6_stride_0, update = value_states_19, x = coreml_update_state_32)[name = string("model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_89_write_state")]; + tensor coreml_update_state_33 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_89")]; + tensor var_2168_begin_0 = const()[name = string("op_2168_begin_0"), val = tensor([2, 0, 0, 0])]; + tensor var_2168_end_0 = const()[name = string("op_2168_end_0"), val = tensor([3, 8, 1024, 128])]; + tensor var_2168_end_mask_0 = const()[name = string("op_2168_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2168_cast_fp16 = slice_by_index(begin = var_2168_begin_0, end = var_2168_end_0, end_mask = var_2168_end_mask_0, x = coreml_update_state_33)[name = string("op_2168_cast_fp16")]; + tensor K_layer_cache_5_axes_0 = const()[name = string("K_layer_cache_5_axes_0"), val = tensor([0])]; + tensor K_layer_cache_5_cast_fp16 = squeeze(axes = K_layer_cache_5_axes_0, x = var_2168_cast_fp16)[name = string("K_layer_cache_5_cast_fp16")]; + tensor var_2175_begin_0 = const()[name = string("op_2175_begin_0"), val = tensor([30, 0, 0, 0])]; + tensor var_2175_end_0 = const()[name = string("op_2175_end_0"), val = tensor([31, 8, 1024, 128])]; + tensor var_2175_end_mask_0 = const()[name = string("op_2175_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2175_cast_fp16 = slice_by_index(begin = var_2175_begin_0, end = var_2175_end_0, end_mask = var_2175_end_mask_0, x = coreml_update_state_33)[name = string("op_2175_cast_fp16")]; + tensor V_layer_cache_5_axes_0 = const()[name = string("V_layer_cache_5_axes_0"), val = tensor([0])]; + tensor V_layer_cache_5_cast_fp16 = squeeze(axes = V_layer_cache_5_axes_0, x = var_2175_cast_fp16)[name = string("V_layer_cache_5_cast_fp16")]; + tensor x_35_axes_0 = const()[name = string("x_35_axes_0"), val = tensor([1])]; + tensor x_35_cast_fp16 = expand_dims(axes = x_35_axes_0, x = K_layer_cache_5_cast_fp16)[name = string("x_35_cast_fp16")]; + tensor var_2204 = const()[name = string("op_2204"), val = tensor([1, 2, 1, 1])]; + tensor x_37_cast_fp16 = tile(reps = var_2204, x = x_35_cast_fp16)[name = string("x_37_cast_fp16")]; + tensor var_2216 = const()[name = string("op_2216"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_27_cast_fp16 = reshape(shape = var_2216, x = x_37_cast_fp16)[name = string("key_states_27_cast_fp16")]; + tensor x_41_axes_0 = const()[name = string("x_41_axes_0"), val = tensor([1])]; + tensor x_41_cast_fp16 = expand_dims(axes = x_41_axes_0, x = V_layer_cache_5_cast_fp16)[name = string("x_41_cast_fp16")]; + tensor var_2224 = const()[name = string("op_2224"), val = tensor([1, 2, 1, 1])]; + tensor x_43_cast_fp16 = tile(reps = var_2224, x = x_41_cast_fp16)[name = string("x_43_cast_fp16")]; + bool var_2251_transpose_x_0 = const()[name = string("op_2251_transpose_x_0"), val = bool(false)]; + bool var_2251_transpose_y_0 = const()[name = string("op_2251_transpose_y_0"), val = bool(true)]; + tensor var_2251 = matmul(transpose_x = var_2251_transpose_x_0, transpose_y = var_2251_transpose_y_0, x = query_states_19, y = key_states_27_cast_fp16)[name = string("op_2251")]; + fp16 var_2252_to_fp16 = const()[name = string("op_2252_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_9_cast_fp16 = mul(x = var_2251, y = var_2252_to_fp16)[name = string("attn_weights_9_cast_fp16")]; + tensor attn_weights_11_cast_fp16 = add(x = attn_weights_9_cast_fp16, y = causal_mask)[name = string("attn_weights_11_cast_fp16")]; + int32 var_2287 = const()[name = string("op_2287"), val = int32(-1)]; + tensor var_2289_cast_fp16 = softmax(axis = var_2287, x = attn_weights_11_cast_fp16)[name = string("op_2289_cast_fp16")]; + tensor concat_48 = const()[name = string("concat_48"), val = tensor([16, 128, 1024])]; + tensor reshape_6_cast_fp16 = reshape(shape = concat_48, x = var_2289_cast_fp16)[name = string("reshape_6_cast_fp16")]; + tensor concat_49 = const()[name = string("concat_49"), val = tensor([16, 1024, 128])]; + tensor reshape_7_cast_fp16 = reshape(shape = concat_49, x = x_43_cast_fp16)[name = string("reshape_7_cast_fp16")]; + bool matmul_2_transpose_x_0 = const()[name = string("matmul_2_transpose_x_0"), val = bool(false)]; + bool matmul_2_transpose_y_0 = const()[name = string("matmul_2_transpose_y_0"), val = bool(false)]; + tensor matmul_2_cast_fp16 = matmul(transpose_x = matmul_2_transpose_x_0, transpose_y = matmul_2_transpose_y_0, x = reshape_6_cast_fp16, y = reshape_7_cast_fp16)[name = string("matmul_2_cast_fp16")]; + tensor concat_53 = const()[name = string("concat_53"), val = tensor([1, 16, 128, 128])]; + tensor reshape_8_cast_fp16 = reshape(shape = concat_53, x = matmul_2_cast_fp16)[name = string("reshape_8_cast_fp16")]; + tensor var_2301_perm_0 = const()[name = string("op_2301_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_2320 = const()[name = string("op_2320"), val = tensor([1, 128, 2048])]; + tensor var_2301_cast_fp16 = transpose(perm = var_2301_perm_0, x = reshape_8_cast_fp16)[name = string("transpose_103")]; + tensor attn_output_25_cast_fp16 = reshape(shape = var_2320, x = var_2301_cast_fp16)[name = string("attn_output_25_cast_fp16")]; + tensor var_2325 = const()[name = string("op_2325"), val = tensor([0, 2, 1])]; + string var_2341_pad_type_0 = const()[name = string("op_2341_pad_type_0"), val = string("valid")]; + int32 var_2341_groups_0 = const()[name = string("op_2341_groups_0"), val = int32(1)]; + tensor var_2341_strides_0 = const()[name = string("op_2341_strides_0"), val = tensor([1])]; + tensor var_2341_pad_0 = const()[name = string("op_2341_pad_0"), val = tensor([0, 0])]; + tensor var_2341_dilations_0 = const()[name = string("op_2341_dilations_0"), val = tensor([1])]; + tensor squeeze_2_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672171136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676365504))))[name = string("squeeze_2_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_2326_cast_fp16 = transpose(perm = var_2325, x = attn_output_25_cast_fp16)[name = string("transpose_102")]; + tensor var_2341_cast_fp16 = conv(dilations = var_2341_dilations_0, groups = var_2341_groups_0, pad = var_2341_pad_0, pad_type = var_2341_pad_type_0, strides = var_2341_strides_0, weight = squeeze_2_cast_fp16_to_fp32_to_fp16_palettized, x = var_2326_cast_fp16)[name = string("op_2341_cast_fp16")]; + tensor var_2345 = const()[name = string("op_2345"), val = tensor([0, 2, 1])]; + tensor attn_output_29_cast_fp16 = transpose(perm = var_2345, x = var_2341_cast_fp16)[name = string("transpose_101")]; + tensor hidden_states_29_cast_fp16 = add(x = hidden_states_21_cast_fp16, y = attn_output_29_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; + int32 var_2358 = const()[name = string("op_2358"), val = int32(-1)]; + fp16 const_99_promoted_to_fp16 = const()[name = string("const_99_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2360_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_99_promoted_to_fp16)[name = string("op_2360_cast_fp16")]; + bool input_47_interleave_0 = const()[name = string("input_47_interleave_0"), val = bool(false)]; + tensor input_47_cast_fp16 = concat(axis = var_2358, interleave = input_47_interleave_0, values = (hidden_states_29_cast_fp16, var_2360_cast_fp16))[name = string("input_47_cast_fp16")]; + tensor normed_45_axes_0 = const()[name = string("normed_45_axes_0"), val = tensor([-1])]; + fp16 var_2355_to_fp16 = const()[name = string("op_2355_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_45_cast_fp16 = layer_norm(axes = normed_45_axes_0, epsilon = var_2355_to_fp16, x = input_47_cast_fp16)[name = string("normed_45_cast_fp16")]; + tensor normed_47_begin_0 = const()[name = string("normed_47_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_47_end_0 = const()[name = string("normed_47_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_47_end_mask_0 = const()[name = string("normed_47_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_47_cast_fp16 = slice_by_index(begin = normed_47_begin_0, end = normed_47_end_0, end_mask = normed_47_end_mask_0, x = normed_45_cast_fp16)[name = string("normed_47_cast_fp16")]; + tensor const_102_promoted_to_fp16 = const()[name = string("const_102_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676496640)))]; + tensor x_45_cast_fp16 = mul(x = normed_47_cast_fp16, y = const_102_promoted_to_fp16)[name = string("x_45_cast_fp16")]; + tensor var_2385 = const()[name = string("op_2385"), val = tensor([0, 2, 1])]; + tensor input_49_axes_0 = const()[name = string("input_49_axes_0"), val = tensor([2])]; + tensor var_2386 = transpose(perm = var_2385, x = x_45_cast_fp16)[name = string("transpose_100")]; + tensor input_49 = expand_dims(axes = input_49_axes_0, x = var_2386)[name = string("input_49")]; + string input_51_pad_type_0 = const()[name = string("input_51_pad_type_0"), val = string("valid")]; + tensor input_51_strides_0 = const()[name = string("input_51_strides_0"), val = tensor([1, 1])]; + tensor input_51_pad_0 = const()[name = string("input_51_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_51_dilations_0 = const()[name = string("input_51_dilations_0"), val = tensor([1, 1])]; + int32 input_51_groups_0 = const()[name = string("input_51_groups_0"), val = int32(1)]; + tensor input_51 = conv(dilations = input_51_dilations_0, groups = input_51_groups_0, pad = input_51_pad_0, pad_type = input_51_pad_type_0, strides = input_51_strides_0, weight = model_model_layers_2_mlp_gate_proj_weight_palettized, x = input_49)[name = string("input_51")]; + string b_5_pad_type_0 = const()[name = string("b_5_pad_type_0"), val = string("valid")]; + tensor b_5_strides_0 = const()[name = string("b_5_strides_0"), val = tensor([1, 1])]; + tensor b_5_pad_0 = const()[name = string("b_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_5_dilations_0 = const()[name = string("b_5_dilations_0"), val = tensor([1, 1])]; + int32 b_5_groups_0 = const()[name = string("b_5_groups_0"), val = int32(1)]; + tensor b_5 = conv(dilations = b_5_dilations_0, groups = b_5_groups_0, pad = b_5_pad_0, pad_type = b_5_pad_type_0, strides = b_5_strides_0, weight = model_model_layers_2_mlp_up_proj_weight_palettized, x = input_49)[name = string("b_5")]; + tensor c_5 = silu(x = input_51)[name = string("c_5")]; + tensor input_53 = mul(x = c_5, y = b_5)[name = string("input_53")]; + string e_5_pad_type_0 = const()[name = string("e_5_pad_type_0"), val = string("valid")]; + tensor e_5_strides_0 = const()[name = string("e_5_strides_0"), val = tensor([1, 1])]; + tensor e_5_pad_0 = const()[name = string("e_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_5_dilations_0 = const()[name = string("e_5_dilations_0"), val = tensor([1, 1])]; + int32 e_5_groups_0 = const()[name = string("e_5_groups_0"), val = int32(1)]; + tensor e_5 = conv(dilations = e_5_dilations_0, groups = e_5_groups_0, pad = e_5_pad_0, pad_type = e_5_pad_type_0, strides = e_5_strides_0, weight = model_model_layers_2_mlp_down_proj_weight_palettized, x = input_53)[name = string("e_5")]; + tensor var_2408_axes_0 = const()[name = string("op_2408_axes_0"), val = tensor([2])]; + tensor var_2408 = squeeze(axes = var_2408_axes_0, x = e_5)[name = string("op_2408")]; + tensor var_2409 = const()[name = string("op_2409"), val = tensor([0, 2, 1])]; + tensor var_2410 = transpose(perm = var_2409, x = var_2408)[name = string("transpose_99")]; + tensor hidden_states_31_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = var_2410)[name = string("hidden_states_31_cast_fp16")]; + int32 var_2422 = const()[name = string("op_2422"), val = int32(-1)]; + fp16 const_103_promoted_to_fp16 = const()[name = string("const_103_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2424_cast_fp16 = mul(x = hidden_states_31_cast_fp16, y = const_103_promoted_to_fp16)[name = string("op_2424_cast_fp16")]; + bool input_55_interleave_0 = const()[name = string("input_55_interleave_0"), val = bool(false)]; + tensor input_55_cast_fp16 = concat(axis = var_2422, interleave = input_55_interleave_0, values = (hidden_states_31_cast_fp16, var_2424_cast_fp16))[name = string("input_55_cast_fp16")]; + tensor normed_49_axes_0 = const()[name = string("normed_49_axes_0"), val = tensor([-1])]; + fp16 var_2419_to_fp16 = const()[name = string("op_2419_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_49_cast_fp16 = layer_norm(axes = normed_49_axes_0, epsilon = var_2419_to_fp16, x = input_55_cast_fp16)[name = string("normed_49_cast_fp16")]; + tensor normed_51_begin_0 = const()[name = string("normed_51_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_51_end_0 = const()[name = string("normed_51_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_51_end_mask_0 = const()[name = string("normed_51_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_51_cast_fp16 = slice_by_index(begin = normed_51_begin_0, end = normed_51_end_0, end_mask = normed_51_end_mask_0, x = normed_49_cast_fp16)[name = string("normed_51_cast_fp16")]; + tensor const_106_promoted_to_fp16 = const()[name = string("const_106_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676500800)))]; + tensor hidden_states_33_cast_fp16 = mul(x = normed_51_cast_fp16, y = const_106_promoted_to_fp16)[name = string("hidden_states_33_cast_fp16")]; + tensor var_2447 = const()[name = string("op_2447"), val = tensor([0, 2, 1])]; + tensor var_2450_axes_0 = const()[name = string("op_2450_axes_0"), val = tensor([2])]; + tensor var_2448_cast_fp16 = transpose(perm = var_2447, x = hidden_states_33_cast_fp16)[name = string("transpose_98")]; + tensor var_2450_cast_fp16 = expand_dims(axes = var_2450_axes_0, x = var_2448_cast_fp16)[name = string("op_2450_cast_fp16")]; + string query_states_25_pad_type_0 = const()[name = string("query_states_25_pad_type_0"), val = string("valid")]; + tensor query_states_25_strides_0 = const()[name = string("query_states_25_strides_0"), val = tensor([1, 1])]; + tensor query_states_25_pad_0 = const()[name = string("query_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_25_dilations_0 = const()[name = string("query_states_25_dilations_0"), val = tensor([1, 1])]; + int32 query_states_25_groups_0 = const()[name = string("query_states_25_groups_0"), val = int32(1)]; + tensor query_states_25 = conv(dilations = query_states_25_dilations_0, groups = query_states_25_groups_0, pad = query_states_25_pad_0, pad_type = query_states_25_pad_type_0, strides = query_states_25_strides_0, weight = model_model_layers_3_self_attn_q_proj_weight_palettized, x = var_2450_cast_fp16)[name = string("query_states_25")]; + string key_states_31_pad_type_0 = const()[name = string("key_states_31_pad_type_0"), val = string("valid")]; + tensor key_states_31_strides_0 = const()[name = string("key_states_31_strides_0"), val = tensor([1, 1])]; + tensor key_states_31_pad_0 = const()[name = string("key_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_31_dilations_0 = const()[name = string("key_states_31_dilations_0"), val = tensor([1, 1])]; + int32 key_states_31_groups_0 = const()[name = string("key_states_31_groups_0"), val = int32(1)]; + tensor key_states_31 = conv(dilations = key_states_31_dilations_0, groups = key_states_31_groups_0, pad = key_states_31_pad_0, pad_type = key_states_31_pad_type_0, strides = key_states_31_strides_0, weight = model_model_layers_3_self_attn_k_proj_weight_palettized, x = var_2450_cast_fp16)[name = string("key_states_31")]; + string value_states_25_pad_type_0 = const()[name = string("value_states_25_pad_type_0"), val = string("valid")]; + tensor value_states_25_strides_0 = const()[name = string("value_states_25_strides_0"), val = tensor([1, 1])]; + tensor value_states_25_pad_0 = const()[name = string("value_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_25_dilations_0 = const()[name = string("value_states_25_dilations_0"), val = tensor([1, 1])]; + int32 value_states_25_groups_0 = const()[name = string("value_states_25_groups_0"), val = int32(1)]; + tensor value_states_25 = conv(dilations = value_states_25_dilations_0, groups = value_states_25_groups_0, pad = value_states_25_pad_0, pad_type = value_states_25_pad_type_0, strides = value_states_25_strides_0, weight = model_model_layers_3_self_attn_v_proj_weight_palettized, x = var_2450_cast_fp16)[name = string("value_states_25")]; + tensor var_2492 = const()[name = string("op_2492"), val = tensor([1, 16, 128, 128])]; + tensor var_2493 = reshape(shape = var_2492, x = query_states_25)[name = string("op_2493")]; + tensor var_2498 = const()[name = string("op_2498"), val = tensor([0, 1, 3, 2])]; + tensor var_2503 = const()[name = string("op_2503"), val = tensor([1, 8, 128, 128])]; + tensor var_2504 = reshape(shape = var_2503, x = key_states_31)[name = string("op_2504")]; + tensor var_2509 = const()[name = string("op_2509"), val = tensor([0, 1, 3, 2])]; + tensor var_2514 = const()[name = string("op_2514"), val = tensor([1, 8, 128, 128])]; + tensor var_2515 = reshape(shape = var_2514, x = value_states_25)[name = string("op_2515")]; + tensor var_2520 = const()[name = string("op_2520"), val = tensor([0, 1, 3, 2])]; + int32 var_2531 = const()[name = string("op_2531"), val = int32(-1)]; + fp16 const_108_promoted = const()[name = string("const_108_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_35 = transpose(perm = var_2498, x = var_2493)[name = string("transpose_97")]; + tensor var_2533 = mul(x = hidden_states_35, y = const_108_promoted)[name = string("op_2533")]; + bool input_59_interleave_0 = const()[name = string("input_59_interleave_0"), val = bool(false)]; + tensor input_59 = concat(axis = var_2531, interleave = input_59_interleave_0, values = (hidden_states_35, var_2533))[name = string("input_59")]; + tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; + fp16 var_2528_to_fp16 = const()[name = string("op_2528_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_2528_to_fp16, x = input_59)[name = string("normed_53_cast_fp16")]; + tensor normed_55_begin_0 = const()[name = string("normed_55_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_55_end_0 = const()[name = string("normed_55_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_55_end_mask_0 = const()[name = string("normed_55_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_55 = slice_by_index(begin = normed_55_begin_0, end = normed_55_end_0, end_mask = normed_55_end_mask_0, x = normed_53_cast_fp16)[name = string("normed_55")]; + tensor const_111 = const()[name = string("const_111"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676504960)))]; + tensor q_7 = mul(x = normed_55, y = const_111)[name = string("q_7")]; + int32 var_2556 = const()[name = string("op_2556"), val = int32(-1)]; + fp16 const_112_promoted = const()[name = string("const_112_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_37 = transpose(perm = var_2509, x = var_2504)[name = string("transpose_96")]; + tensor var_2558 = mul(x = hidden_states_37, y = const_112_promoted)[name = string("op_2558")]; + bool input_61_interleave_0 = const()[name = string("input_61_interleave_0"), val = bool(false)]; + tensor input_61 = concat(axis = var_2556, interleave = input_61_interleave_0, values = (hidden_states_37, var_2558))[name = string("input_61")]; + tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; + fp16 var_2553_to_fp16 = const()[name = string("op_2553_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_2553_to_fp16, x = input_61)[name = string("normed_57_cast_fp16")]; + tensor normed_59_begin_0 = const()[name = string("normed_59_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_59_end_0 = const()[name = string("normed_59_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_59_end_mask_0 = const()[name = string("normed_59_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_59 = slice_by_index(begin = normed_59_begin_0, end = normed_59_end_0, end_mask = normed_59_end_mask_0, x = normed_57_cast_fp16)[name = string("normed_59")]; + tensor const_115 = const()[name = string("const_115"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676505280)))]; + tensor k_7 = mul(x = normed_59, y = const_115)[name = string("k_7")]; + tensor var_2584 = mul(x = q_7, y = cos_5)[name = string("op_2584")]; + tensor x1_13_begin_0 = const()[name = string("x1_13_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_13_end_0 = const()[name = string("x1_13_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_13_end_mask_0 = const()[name = string("x1_13_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_13 = slice_by_index(begin = x1_13_begin_0, end = x1_13_end_0, end_mask = x1_13_end_mask_0, x = q_7)[name = string("x1_13")]; + tensor x2_13_begin_0 = const()[name = string("x2_13_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_13_end_0 = const()[name = string("x2_13_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_13_end_mask_0 = const()[name = string("x2_13_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_13 = slice_by_index(begin = x2_13_begin_0, end = x2_13_end_0, end_mask = x2_13_end_mask_0, x = q_7)[name = string("x2_13")]; + fp16 const_118_promoted = const()[name = string("const_118_promoted"), val = fp16(-0x1p+0)]; + tensor var_2605 = mul(x = x2_13, y = const_118_promoted)[name = string("op_2605")]; + int32 var_2607 = const()[name = string("op_2607"), val = int32(-1)]; + bool var_2608_interleave_0 = const()[name = string("op_2608_interleave_0"), val = bool(false)]; + tensor var_2608 = concat(axis = var_2607, interleave = var_2608_interleave_0, values = (var_2605, x1_13))[name = string("op_2608")]; + tensor var_2609 = mul(x = var_2608, y = sin_5)[name = string("op_2609")]; + tensor query_states_27 = add(x = var_2584, y = var_2609)[name = string("query_states_27")]; + tensor var_2612 = mul(x = k_7, y = cos_5)[name = string("op_2612")]; + tensor x1_15_begin_0 = const()[name = string("x1_15_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_15_end_0 = const()[name = string("x1_15_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_15_end_mask_0 = const()[name = string("x1_15_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_15 = slice_by_index(begin = x1_15_begin_0, end = x1_15_end_0, end_mask = x1_15_end_mask_0, x = k_7)[name = string("x1_15")]; + tensor x2_15_begin_0 = const()[name = string("x2_15_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_15_end_0 = const()[name = string("x2_15_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_15_end_mask_0 = const()[name = string("x2_15_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_15 = slice_by_index(begin = x2_15_begin_0, end = x2_15_end_0, end_mask = x2_15_end_mask_0, x = k_7)[name = string("x2_15")]; + fp16 const_121_promoted = const()[name = string("const_121_promoted"), val = fp16(-0x1p+0)]; + tensor var_2633 = mul(x = x2_15, y = const_121_promoted)[name = string("op_2633")]; + int32 var_2635 = const()[name = string("op_2635"), val = int32(-1)]; + bool var_2636_interleave_0 = const()[name = string("op_2636_interleave_0"), val = bool(false)]; + tensor var_2636 = concat(axis = var_2635, interleave = var_2636_interleave_0, values = (var_2633, x1_15))[name = string("op_2636")]; + tensor var_2637 = mul(x = var_2636, y = sin_5)[name = string("op_2637")]; + tensor key_states_33 = add(x = var_2612, y = var_2637)[name = string("key_states_33")]; + tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([3])]; + tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; + tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; + tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([4])]; + int32 concat_56_axis_0 = const()[name = string("concat_56_axis_0"), val = int32(0)]; + bool concat_56_interleave_0 = const()[name = string("concat_56_interleave_0"), val = bool(false)]; + tensor concat_56 = concat(axis = concat_56_axis_0, interleave = concat_56_interleave_0, values = (expand_dims_36, expand_dims_37, current_pos, expand_dims_39))[name = string("concat_56")]; + tensor concat_57_values1_0 = const()[name = string("concat_57_values1_0"), val = tensor([0])]; + tensor concat_57_values3_0 = const()[name = string("concat_57_values3_0"), val = tensor([0])]; + int32 concat_57_axis_0 = const()[name = string("concat_57_axis_0"), val = int32(0)]; + bool concat_57_interleave_0 = const()[name = string("concat_57_interleave_0"), val = bool(false)]; + tensor concat_57 = concat(axis = concat_57_axis_0, interleave = concat_57_interleave_0, values = (expand_dims_40, concat_57_values1_0, var_1039, concat_57_values3_0))[name = string("concat_57")]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_56, begin_mask = model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0, end = concat_57, end_mask = model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_7_stride_0, update = key_states_33, x = coreml_update_state_33)[name = string("model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_90_write_state")]; + tensor coreml_update_state_34 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_90")]; + tensor expand_dims_42 = const()[name = string("expand_dims_42"), val = tensor([31])]; + tensor expand_dims_43 = const()[name = string("expand_dims_43"), val = tensor([0])]; + tensor expand_dims_45 = const()[name = string("expand_dims_45"), val = tensor([0])]; + tensor expand_dims_46 = const()[name = string("expand_dims_46"), val = tensor([32])]; + int32 concat_60_axis_0 = const()[name = string("concat_60_axis_0"), val = int32(0)]; + bool concat_60_interleave_0 = const()[name = string("concat_60_interleave_0"), val = bool(false)]; + tensor concat_60 = concat(axis = concat_60_axis_0, interleave = concat_60_interleave_0, values = (expand_dims_42, expand_dims_43, current_pos, expand_dims_45))[name = string("concat_60")]; + tensor concat_61_values1_0 = const()[name = string("concat_61_values1_0"), val = tensor([0])]; + tensor concat_61_values3_0 = const()[name = string("concat_61_values3_0"), val = tensor([0])]; + int32 concat_61_axis_0 = const()[name = string("concat_61_axis_0"), val = int32(0)]; + bool concat_61_interleave_0 = const()[name = string("concat_61_interleave_0"), val = bool(false)]; + tensor concat_61 = concat(axis = concat_61_axis_0, interleave = concat_61_interleave_0, values = (expand_dims_46, concat_61_values1_0, var_1039, concat_61_values3_0))[name = string("concat_61")]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_27 = transpose(perm = var_2520, x = var_2515)[name = string("transpose_95")]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_60, begin_mask = model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0, end = concat_61, end_mask = model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_8_stride_0, update = value_states_27, x = coreml_update_state_34)[name = string("model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_91_write_state")]; + tensor coreml_update_state_35 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_91")]; + tensor var_2708_begin_0 = const()[name = string("op_2708_begin_0"), val = tensor([3, 0, 0, 0])]; + tensor var_2708_end_0 = const()[name = string("op_2708_end_0"), val = tensor([4, 8, 1024, 128])]; + tensor var_2708_end_mask_0 = const()[name = string("op_2708_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2708_cast_fp16 = slice_by_index(begin = var_2708_begin_0, end = var_2708_end_0, end_mask = var_2708_end_mask_0, x = coreml_update_state_35)[name = string("op_2708_cast_fp16")]; + tensor K_layer_cache_7_axes_0 = const()[name = string("K_layer_cache_7_axes_0"), val = tensor([0])]; + tensor K_layer_cache_7_cast_fp16 = squeeze(axes = K_layer_cache_7_axes_0, x = var_2708_cast_fp16)[name = string("K_layer_cache_7_cast_fp16")]; + tensor var_2715_begin_0 = const()[name = string("op_2715_begin_0"), val = tensor([31, 0, 0, 0])]; + tensor var_2715_end_0 = const()[name = string("op_2715_end_0"), val = tensor([32, 8, 1024, 128])]; + tensor var_2715_end_mask_0 = const()[name = string("op_2715_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2715_cast_fp16 = slice_by_index(begin = var_2715_begin_0, end = var_2715_end_0, end_mask = var_2715_end_mask_0, x = coreml_update_state_35)[name = string("op_2715_cast_fp16")]; + tensor V_layer_cache_7_axes_0 = const()[name = string("V_layer_cache_7_axes_0"), val = tensor([0])]; + tensor V_layer_cache_7_cast_fp16 = squeeze(axes = V_layer_cache_7_axes_0, x = var_2715_cast_fp16)[name = string("V_layer_cache_7_cast_fp16")]; + tensor x_51_axes_0 = const()[name = string("x_51_axes_0"), val = tensor([1])]; + tensor x_51_cast_fp16 = expand_dims(axes = x_51_axes_0, x = K_layer_cache_7_cast_fp16)[name = string("x_51_cast_fp16")]; + tensor var_2744 = const()[name = string("op_2744"), val = tensor([1, 2, 1, 1])]; + tensor x_53_cast_fp16 = tile(reps = var_2744, x = x_51_cast_fp16)[name = string("x_53_cast_fp16")]; + tensor var_2756 = const()[name = string("op_2756"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_37_cast_fp16 = reshape(shape = var_2756, x = x_53_cast_fp16)[name = string("key_states_37_cast_fp16")]; + tensor x_57_axes_0 = const()[name = string("x_57_axes_0"), val = tensor([1])]; + tensor x_57_cast_fp16 = expand_dims(axes = x_57_axes_0, x = V_layer_cache_7_cast_fp16)[name = string("x_57_cast_fp16")]; + tensor var_2764 = const()[name = string("op_2764"), val = tensor([1, 2, 1, 1])]; + tensor x_59_cast_fp16 = tile(reps = var_2764, x = x_57_cast_fp16)[name = string("x_59_cast_fp16")]; + bool var_2791_transpose_x_0 = const()[name = string("op_2791_transpose_x_0"), val = bool(false)]; + bool var_2791_transpose_y_0 = const()[name = string("op_2791_transpose_y_0"), val = bool(true)]; + tensor var_2791 = matmul(transpose_x = var_2791_transpose_x_0, transpose_y = var_2791_transpose_y_0, x = query_states_27, y = key_states_37_cast_fp16)[name = string("op_2791")]; + fp16 var_2792_to_fp16 = const()[name = string("op_2792_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_13_cast_fp16 = mul(x = var_2791, y = var_2792_to_fp16)[name = string("attn_weights_13_cast_fp16")]; + tensor attn_weights_15_cast_fp16 = add(x = attn_weights_13_cast_fp16, y = causal_mask)[name = string("attn_weights_15_cast_fp16")]; + int32 var_2827 = const()[name = string("op_2827"), val = int32(-1)]; + tensor var_2829_cast_fp16 = softmax(axis = var_2827, x = attn_weights_15_cast_fp16)[name = string("op_2829_cast_fp16")]; + tensor concat_66 = const()[name = string("concat_66"), val = tensor([16, 128, 1024])]; + tensor reshape_9_cast_fp16 = reshape(shape = concat_66, x = var_2829_cast_fp16)[name = string("reshape_9_cast_fp16")]; + tensor concat_67 = const()[name = string("concat_67"), val = tensor([16, 1024, 128])]; + tensor reshape_10_cast_fp16 = reshape(shape = concat_67, x = x_59_cast_fp16)[name = string("reshape_10_cast_fp16")]; + bool matmul_3_transpose_x_0 = const()[name = string("matmul_3_transpose_x_0"), val = bool(false)]; + bool matmul_3_transpose_y_0 = const()[name = string("matmul_3_transpose_y_0"), val = bool(false)]; + tensor matmul_3_cast_fp16 = matmul(transpose_x = matmul_3_transpose_x_0, transpose_y = matmul_3_transpose_y_0, x = reshape_9_cast_fp16, y = reshape_10_cast_fp16)[name = string("matmul_3_cast_fp16")]; + tensor concat_71 = const()[name = string("concat_71"), val = tensor([1, 16, 128, 128])]; + tensor reshape_11_cast_fp16 = reshape(shape = concat_71, x = matmul_3_cast_fp16)[name = string("reshape_11_cast_fp16")]; + tensor var_2841_perm_0 = const()[name = string("op_2841_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_2860 = const()[name = string("op_2860"), val = tensor([1, 128, 2048])]; + tensor var_2841_cast_fp16 = transpose(perm = var_2841_perm_0, x = reshape_11_cast_fp16)[name = string("transpose_94")]; + tensor attn_output_35_cast_fp16 = reshape(shape = var_2860, x = var_2841_cast_fp16)[name = string("attn_output_35_cast_fp16")]; + tensor var_2865 = const()[name = string("op_2865"), val = tensor([0, 2, 1])]; + string var_2881_pad_type_0 = const()[name = string("op_2881_pad_type_0"), val = string("valid")]; + int32 var_2881_groups_0 = const()[name = string("op_2881_groups_0"), val = int32(1)]; + tensor var_2881_strides_0 = const()[name = string("op_2881_strides_0"), val = tensor([1])]; + tensor var_2881_pad_0 = const()[name = string("op_2881_pad_0"), val = tensor([0, 0])]; + tensor var_2881_dilations_0 = const()[name = string("op_2881_dilations_0"), val = tensor([1])]; + tensor squeeze_3_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676505600))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680699968))))[name = string("squeeze_3_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_2866_cast_fp16 = transpose(perm = var_2865, x = attn_output_35_cast_fp16)[name = string("transpose_93")]; + tensor var_2881_cast_fp16 = conv(dilations = var_2881_dilations_0, groups = var_2881_groups_0, pad = var_2881_pad_0, pad_type = var_2881_pad_type_0, strides = var_2881_strides_0, weight = squeeze_3_cast_fp16_to_fp32_to_fp16_palettized, x = var_2866_cast_fp16)[name = string("op_2881_cast_fp16")]; + tensor var_2885 = const()[name = string("op_2885"), val = tensor([0, 2, 1])]; + tensor attn_output_39_cast_fp16 = transpose(perm = var_2885, x = var_2881_cast_fp16)[name = string("transpose_92")]; + tensor hidden_states_39_cast_fp16 = add(x = hidden_states_31_cast_fp16, y = attn_output_39_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; + int32 var_2898 = const()[name = string("op_2898"), val = int32(-1)]; + fp16 const_133_promoted_to_fp16 = const()[name = string("const_133_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2900_cast_fp16 = mul(x = hidden_states_39_cast_fp16, y = const_133_promoted_to_fp16)[name = string("op_2900_cast_fp16")]; + bool input_65_interleave_0 = const()[name = string("input_65_interleave_0"), val = bool(false)]; + tensor input_65_cast_fp16 = concat(axis = var_2898, interleave = input_65_interleave_0, values = (hidden_states_39_cast_fp16, var_2900_cast_fp16))[name = string("input_65_cast_fp16")]; + tensor normed_61_axes_0 = const()[name = string("normed_61_axes_0"), val = tensor([-1])]; + fp16 var_2895_to_fp16 = const()[name = string("op_2895_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_61_cast_fp16 = layer_norm(axes = normed_61_axes_0, epsilon = var_2895_to_fp16, x = input_65_cast_fp16)[name = string("normed_61_cast_fp16")]; + tensor normed_63_begin_0 = const()[name = string("normed_63_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_63_end_0 = const()[name = string("normed_63_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_63_end_mask_0 = const()[name = string("normed_63_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_63_cast_fp16 = slice_by_index(begin = normed_63_begin_0, end = normed_63_end_0, end_mask = normed_63_end_mask_0, x = normed_61_cast_fp16)[name = string("normed_63_cast_fp16")]; + tensor const_136_promoted_to_fp16 = const()[name = string("const_136_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680831104)))]; + tensor x_61_cast_fp16 = mul(x = normed_63_cast_fp16, y = const_136_promoted_to_fp16)[name = string("x_61_cast_fp16")]; + tensor var_2925 = const()[name = string("op_2925"), val = tensor([0, 2, 1])]; + tensor input_67_axes_0 = const()[name = string("input_67_axes_0"), val = tensor([2])]; + tensor var_2926 = transpose(perm = var_2925, x = x_61_cast_fp16)[name = string("transpose_91")]; + tensor input_67 = expand_dims(axes = input_67_axes_0, x = var_2926)[name = string("input_67")]; + string input_69_pad_type_0 = const()[name = string("input_69_pad_type_0"), val = string("valid")]; + tensor input_69_strides_0 = const()[name = string("input_69_strides_0"), val = tensor([1, 1])]; + tensor input_69_pad_0 = const()[name = string("input_69_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_69_dilations_0 = const()[name = string("input_69_dilations_0"), val = tensor([1, 1])]; + int32 input_69_groups_0 = const()[name = string("input_69_groups_0"), val = int32(1)]; + tensor input_69 = conv(dilations = input_69_dilations_0, groups = input_69_groups_0, pad = input_69_pad_0, pad_type = input_69_pad_type_0, strides = input_69_strides_0, weight = model_model_layers_3_mlp_gate_proj_weight_palettized, x = input_67)[name = string("input_69")]; + string b_7_pad_type_0 = const()[name = string("b_7_pad_type_0"), val = string("valid")]; + tensor b_7_strides_0 = const()[name = string("b_7_strides_0"), val = tensor([1, 1])]; + tensor b_7_pad_0 = const()[name = string("b_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_7_dilations_0 = const()[name = string("b_7_dilations_0"), val = tensor([1, 1])]; + int32 b_7_groups_0 = const()[name = string("b_7_groups_0"), val = int32(1)]; + tensor b_7 = conv(dilations = b_7_dilations_0, groups = b_7_groups_0, pad = b_7_pad_0, pad_type = b_7_pad_type_0, strides = b_7_strides_0, weight = model_model_layers_3_mlp_up_proj_weight_palettized, x = input_67)[name = string("b_7")]; + tensor c_7 = silu(x = input_69)[name = string("c_7")]; + tensor input_71 = mul(x = c_7, y = b_7)[name = string("input_71")]; + string e_7_pad_type_0 = const()[name = string("e_7_pad_type_0"), val = string("valid")]; + tensor e_7_strides_0 = const()[name = string("e_7_strides_0"), val = tensor([1, 1])]; + tensor e_7_pad_0 = const()[name = string("e_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_7_dilations_0 = const()[name = string("e_7_dilations_0"), val = tensor([1, 1])]; + int32 e_7_groups_0 = const()[name = string("e_7_groups_0"), val = int32(1)]; + tensor e_7 = conv(dilations = e_7_dilations_0, groups = e_7_groups_0, pad = e_7_pad_0, pad_type = e_7_pad_type_0, strides = e_7_strides_0, weight = model_model_layers_3_mlp_down_proj_weight_palettized, x = input_71)[name = string("e_7")]; + tensor var_2948_axes_0 = const()[name = string("op_2948_axes_0"), val = tensor([2])]; + tensor var_2948 = squeeze(axes = var_2948_axes_0, x = e_7)[name = string("op_2948")]; + tensor var_2949 = const()[name = string("op_2949"), val = tensor([0, 2, 1])]; + tensor var_2950 = transpose(perm = var_2949, x = var_2948)[name = string("transpose_90")]; + tensor hidden_states_41_cast_fp16 = add(x = hidden_states_39_cast_fp16, y = var_2950)[name = string("hidden_states_41_cast_fp16")]; + int32 var_2962 = const()[name = string("op_2962"), val = int32(-1)]; + fp16 const_137_promoted_to_fp16 = const()[name = string("const_137_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2964_cast_fp16 = mul(x = hidden_states_41_cast_fp16, y = const_137_promoted_to_fp16)[name = string("op_2964_cast_fp16")]; + bool input_73_interleave_0 = const()[name = string("input_73_interleave_0"), val = bool(false)]; + tensor input_73_cast_fp16 = concat(axis = var_2962, interleave = input_73_interleave_0, values = (hidden_states_41_cast_fp16, var_2964_cast_fp16))[name = string("input_73_cast_fp16")]; + tensor normed_65_axes_0 = const()[name = string("normed_65_axes_0"), val = tensor([-1])]; + fp16 var_2959_to_fp16 = const()[name = string("op_2959_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_65_cast_fp16 = layer_norm(axes = normed_65_axes_0, epsilon = var_2959_to_fp16, x = input_73_cast_fp16)[name = string("normed_65_cast_fp16")]; + tensor normed_67_begin_0 = const()[name = string("normed_67_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_67_end_0 = const()[name = string("normed_67_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_67_end_mask_0 = const()[name = string("normed_67_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_67_cast_fp16 = slice_by_index(begin = normed_67_begin_0, end = normed_67_end_0, end_mask = normed_67_end_mask_0, x = normed_65_cast_fp16)[name = string("normed_67_cast_fp16")]; + tensor const_140_promoted_to_fp16 = const()[name = string("const_140_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680835264)))]; + tensor hidden_states_43_cast_fp16 = mul(x = normed_67_cast_fp16, y = const_140_promoted_to_fp16)[name = string("hidden_states_43_cast_fp16")]; + tensor var_2987 = const()[name = string("op_2987"), val = tensor([0, 2, 1])]; + tensor var_2990_axes_0 = const()[name = string("op_2990_axes_0"), val = tensor([2])]; + tensor var_2988_cast_fp16 = transpose(perm = var_2987, x = hidden_states_43_cast_fp16)[name = string("transpose_89")]; + tensor var_2990_cast_fp16 = expand_dims(axes = var_2990_axes_0, x = var_2988_cast_fp16)[name = string("op_2990_cast_fp16")]; + string query_states_33_pad_type_0 = const()[name = string("query_states_33_pad_type_0"), val = string("valid")]; + tensor query_states_33_strides_0 = const()[name = string("query_states_33_strides_0"), val = tensor([1, 1])]; + tensor query_states_33_pad_0 = const()[name = string("query_states_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_33_dilations_0 = const()[name = string("query_states_33_dilations_0"), val = tensor([1, 1])]; + int32 query_states_33_groups_0 = const()[name = string("query_states_33_groups_0"), val = int32(1)]; + tensor query_states_33 = conv(dilations = query_states_33_dilations_0, groups = query_states_33_groups_0, pad = query_states_33_pad_0, pad_type = query_states_33_pad_type_0, strides = query_states_33_strides_0, weight = model_model_layers_4_self_attn_q_proj_weight_palettized, x = var_2990_cast_fp16)[name = string("query_states_33")]; + string key_states_41_pad_type_0 = const()[name = string("key_states_41_pad_type_0"), val = string("valid")]; + tensor key_states_41_strides_0 = const()[name = string("key_states_41_strides_0"), val = tensor([1, 1])]; + tensor key_states_41_pad_0 = const()[name = string("key_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_41_dilations_0 = const()[name = string("key_states_41_dilations_0"), val = tensor([1, 1])]; + int32 key_states_41_groups_0 = const()[name = string("key_states_41_groups_0"), val = int32(1)]; + tensor key_states_41 = conv(dilations = key_states_41_dilations_0, groups = key_states_41_groups_0, pad = key_states_41_pad_0, pad_type = key_states_41_pad_type_0, strides = key_states_41_strides_0, weight = model_model_layers_4_self_attn_k_proj_weight_palettized, x = var_2990_cast_fp16)[name = string("key_states_41")]; + string value_states_33_pad_type_0 = const()[name = string("value_states_33_pad_type_0"), val = string("valid")]; + tensor value_states_33_strides_0 = const()[name = string("value_states_33_strides_0"), val = tensor([1, 1])]; + tensor value_states_33_pad_0 = const()[name = string("value_states_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_33_dilations_0 = const()[name = string("value_states_33_dilations_0"), val = tensor([1, 1])]; + int32 value_states_33_groups_0 = const()[name = string("value_states_33_groups_0"), val = int32(1)]; + tensor value_states_33 = conv(dilations = value_states_33_dilations_0, groups = value_states_33_groups_0, pad = value_states_33_pad_0, pad_type = value_states_33_pad_type_0, strides = value_states_33_strides_0, weight = model_model_layers_4_self_attn_v_proj_weight_palettized, x = var_2990_cast_fp16)[name = string("value_states_33")]; + tensor var_3032 = const()[name = string("op_3032"), val = tensor([1, 16, 128, 128])]; + tensor var_3033 = reshape(shape = var_3032, x = query_states_33)[name = string("op_3033")]; + tensor var_3038 = const()[name = string("op_3038"), val = tensor([0, 1, 3, 2])]; + tensor var_3043 = const()[name = string("op_3043"), val = tensor([1, 8, 128, 128])]; + tensor var_3044 = reshape(shape = var_3043, x = key_states_41)[name = string("op_3044")]; + tensor var_3049 = const()[name = string("op_3049"), val = tensor([0, 1, 3, 2])]; + tensor var_3054 = const()[name = string("op_3054"), val = tensor([1, 8, 128, 128])]; + tensor var_3055 = reshape(shape = var_3054, x = value_states_33)[name = string("op_3055")]; + tensor var_3060 = const()[name = string("op_3060"), val = tensor([0, 1, 3, 2])]; + int32 var_3071 = const()[name = string("op_3071"), val = int32(-1)]; + fp16 const_142_promoted = const()[name = string("const_142_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_45 = transpose(perm = var_3038, x = var_3033)[name = string("transpose_88")]; + tensor var_3073 = mul(x = hidden_states_45, y = const_142_promoted)[name = string("op_3073")]; + bool input_77_interleave_0 = const()[name = string("input_77_interleave_0"), val = bool(false)]; + tensor input_77 = concat(axis = var_3071, interleave = input_77_interleave_0, values = (hidden_states_45, var_3073))[name = string("input_77")]; + tensor normed_69_axes_0 = const()[name = string("normed_69_axes_0"), val = tensor([-1])]; + fp16 var_3068_to_fp16 = const()[name = string("op_3068_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_69_cast_fp16 = layer_norm(axes = normed_69_axes_0, epsilon = var_3068_to_fp16, x = input_77)[name = string("normed_69_cast_fp16")]; + tensor normed_71_begin_0 = const()[name = string("normed_71_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_71_end_0 = const()[name = string("normed_71_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_71_end_mask_0 = const()[name = string("normed_71_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_71 = slice_by_index(begin = normed_71_begin_0, end = normed_71_end_0, end_mask = normed_71_end_mask_0, x = normed_69_cast_fp16)[name = string("normed_71")]; + tensor const_145 = const()[name = string("const_145"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680839424)))]; + tensor q_9 = mul(x = normed_71, y = const_145)[name = string("q_9")]; + int32 var_3096 = const()[name = string("op_3096"), val = int32(-1)]; + fp16 const_146_promoted = const()[name = string("const_146_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_47 = transpose(perm = var_3049, x = var_3044)[name = string("transpose_87")]; + tensor var_3098 = mul(x = hidden_states_47, y = const_146_promoted)[name = string("op_3098")]; + bool input_79_interleave_0 = const()[name = string("input_79_interleave_0"), val = bool(false)]; + tensor input_79 = concat(axis = var_3096, interleave = input_79_interleave_0, values = (hidden_states_47, var_3098))[name = string("input_79")]; + tensor normed_73_axes_0 = const()[name = string("normed_73_axes_0"), val = tensor([-1])]; + fp16 var_3093_to_fp16 = const()[name = string("op_3093_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_73_cast_fp16 = layer_norm(axes = normed_73_axes_0, epsilon = var_3093_to_fp16, x = input_79)[name = string("normed_73_cast_fp16")]; + tensor normed_75_begin_0 = const()[name = string("normed_75_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_75_end_0 = const()[name = string("normed_75_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_75_end_mask_0 = const()[name = string("normed_75_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_75 = slice_by_index(begin = normed_75_begin_0, end = normed_75_end_0, end_mask = normed_75_end_mask_0, x = normed_73_cast_fp16)[name = string("normed_75")]; + tensor const_149 = const()[name = string("const_149"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680839744)))]; + tensor k_9 = mul(x = normed_75, y = const_149)[name = string("k_9")]; + tensor var_3124 = mul(x = q_9, y = cos_5)[name = string("op_3124")]; + tensor x1_17_begin_0 = const()[name = string("x1_17_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_17_end_0 = const()[name = string("x1_17_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_17_end_mask_0 = const()[name = string("x1_17_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_17 = slice_by_index(begin = x1_17_begin_0, end = x1_17_end_0, end_mask = x1_17_end_mask_0, x = q_9)[name = string("x1_17")]; + tensor x2_17_begin_0 = const()[name = string("x2_17_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_17_end_0 = const()[name = string("x2_17_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_17_end_mask_0 = const()[name = string("x2_17_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_17 = slice_by_index(begin = x2_17_begin_0, end = x2_17_end_0, end_mask = x2_17_end_mask_0, x = q_9)[name = string("x2_17")]; + fp16 const_152_promoted = const()[name = string("const_152_promoted"), val = fp16(-0x1p+0)]; + tensor var_3145 = mul(x = x2_17, y = const_152_promoted)[name = string("op_3145")]; + int32 var_3147 = const()[name = string("op_3147"), val = int32(-1)]; + bool var_3148_interleave_0 = const()[name = string("op_3148_interleave_0"), val = bool(false)]; + tensor var_3148 = concat(axis = var_3147, interleave = var_3148_interleave_0, values = (var_3145, x1_17))[name = string("op_3148")]; + tensor var_3149 = mul(x = var_3148, y = sin_5)[name = string("op_3149")]; + tensor query_states_35 = add(x = var_3124, y = var_3149)[name = string("query_states_35")]; + tensor var_3152 = mul(x = k_9, y = cos_5)[name = string("op_3152")]; + tensor x1_19_begin_0 = const()[name = string("x1_19_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_19_end_0 = const()[name = string("x1_19_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_19_end_mask_0 = const()[name = string("x1_19_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_19 = slice_by_index(begin = x1_19_begin_0, end = x1_19_end_0, end_mask = x1_19_end_mask_0, x = k_9)[name = string("x1_19")]; + tensor x2_19_begin_0 = const()[name = string("x2_19_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_19_end_0 = const()[name = string("x2_19_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_19_end_mask_0 = const()[name = string("x2_19_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_19 = slice_by_index(begin = x2_19_begin_0, end = x2_19_end_0, end_mask = x2_19_end_mask_0, x = k_9)[name = string("x2_19")]; + fp16 const_155_promoted = const()[name = string("const_155_promoted"), val = fp16(-0x1p+0)]; + tensor var_3173 = mul(x = x2_19, y = const_155_promoted)[name = string("op_3173")]; + int32 var_3175 = const()[name = string("op_3175"), val = int32(-1)]; + bool var_3176_interleave_0 = const()[name = string("op_3176_interleave_0"), val = bool(false)]; + tensor var_3176 = concat(axis = var_3175, interleave = var_3176_interleave_0, values = (var_3173, x1_19))[name = string("op_3176")]; + tensor var_3177 = mul(x = var_3176, y = sin_5)[name = string("op_3177")]; + tensor key_states_43 = add(x = var_3152, y = var_3177)[name = string("key_states_43")]; + tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([4])]; + tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; + tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; + tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([5])]; + int32 concat_74_axis_0 = const()[name = string("concat_74_axis_0"), val = int32(0)]; + bool concat_74_interleave_0 = const()[name = string("concat_74_interleave_0"), val = bool(false)]; + tensor concat_74 = concat(axis = concat_74_axis_0, interleave = concat_74_interleave_0, values = (expand_dims_48, expand_dims_49, current_pos, expand_dims_51))[name = string("concat_74")]; + tensor concat_75_values1_0 = const()[name = string("concat_75_values1_0"), val = tensor([0])]; + tensor concat_75_values3_0 = const()[name = string("concat_75_values3_0"), val = tensor([0])]; + int32 concat_75_axis_0 = const()[name = string("concat_75_axis_0"), val = int32(0)]; + bool concat_75_interleave_0 = const()[name = string("concat_75_interleave_0"), val = bool(false)]; + tensor concat_75 = concat(axis = concat_75_axis_0, interleave = concat_75_interleave_0, values = (expand_dims_52, concat_75_values1_0, var_1039, concat_75_values3_0))[name = string("concat_75")]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_74, begin_mask = model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0, end = concat_75, end_mask = model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_9_stride_0, update = key_states_43, x = coreml_update_state_35)[name = string("model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_92_write_state")]; + tensor coreml_update_state_36 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_92")]; + tensor expand_dims_54 = const()[name = string("expand_dims_54"), val = tensor([32])]; + tensor expand_dims_55 = const()[name = string("expand_dims_55"), val = tensor([0])]; + tensor expand_dims_57 = const()[name = string("expand_dims_57"), val = tensor([0])]; + tensor expand_dims_58 = const()[name = string("expand_dims_58"), val = tensor([33])]; + int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; + bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; + tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_54, expand_dims_55, current_pos, expand_dims_57))[name = string("concat_78")]; + tensor concat_79_values1_0 = const()[name = string("concat_79_values1_0"), val = tensor([0])]; + tensor concat_79_values3_0 = const()[name = string("concat_79_values3_0"), val = tensor([0])]; + int32 concat_79_axis_0 = const()[name = string("concat_79_axis_0"), val = int32(0)]; + bool concat_79_interleave_0 = const()[name = string("concat_79_interleave_0"), val = bool(false)]; + tensor concat_79 = concat(axis = concat_79_axis_0, interleave = concat_79_interleave_0, values = (expand_dims_58, concat_79_values1_0, var_1039, concat_79_values3_0))[name = string("concat_79")]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_35 = transpose(perm = var_3060, x = var_3055)[name = string("transpose_86")]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_78, begin_mask = model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0, end = concat_79, end_mask = model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_10_stride_0, update = value_states_35, x = coreml_update_state_36)[name = string("model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_93_write_state")]; + tensor coreml_update_state_37 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_93")]; + tensor var_3248_begin_0 = const()[name = string("op_3248_begin_0"), val = tensor([4, 0, 0, 0])]; + tensor var_3248_end_0 = const()[name = string("op_3248_end_0"), val = tensor([5, 8, 1024, 128])]; + tensor var_3248_end_mask_0 = const()[name = string("op_3248_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3248_cast_fp16 = slice_by_index(begin = var_3248_begin_0, end = var_3248_end_0, end_mask = var_3248_end_mask_0, x = coreml_update_state_37)[name = string("op_3248_cast_fp16")]; + tensor K_layer_cache_9_axes_0 = const()[name = string("K_layer_cache_9_axes_0"), val = tensor([0])]; + tensor K_layer_cache_9_cast_fp16 = squeeze(axes = K_layer_cache_9_axes_0, x = var_3248_cast_fp16)[name = string("K_layer_cache_9_cast_fp16")]; + tensor var_3255_begin_0 = const()[name = string("op_3255_begin_0"), val = tensor([32, 0, 0, 0])]; + tensor var_3255_end_0 = const()[name = string("op_3255_end_0"), val = tensor([33, 8, 1024, 128])]; + tensor var_3255_end_mask_0 = const()[name = string("op_3255_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3255_cast_fp16 = slice_by_index(begin = var_3255_begin_0, end = var_3255_end_0, end_mask = var_3255_end_mask_0, x = coreml_update_state_37)[name = string("op_3255_cast_fp16")]; + tensor V_layer_cache_9_axes_0 = const()[name = string("V_layer_cache_9_axes_0"), val = tensor([0])]; + tensor V_layer_cache_9_cast_fp16 = squeeze(axes = V_layer_cache_9_axes_0, x = var_3255_cast_fp16)[name = string("V_layer_cache_9_cast_fp16")]; + tensor x_67_axes_0 = const()[name = string("x_67_axes_0"), val = tensor([1])]; + tensor x_67_cast_fp16 = expand_dims(axes = x_67_axes_0, x = K_layer_cache_9_cast_fp16)[name = string("x_67_cast_fp16")]; + tensor var_3284 = const()[name = string("op_3284"), val = tensor([1, 2, 1, 1])]; + tensor x_69_cast_fp16 = tile(reps = var_3284, x = x_67_cast_fp16)[name = string("x_69_cast_fp16")]; + tensor var_3296 = const()[name = string("op_3296"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_47_cast_fp16 = reshape(shape = var_3296, x = x_69_cast_fp16)[name = string("key_states_47_cast_fp16")]; + tensor x_73_axes_0 = const()[name = string("x_73_axes_0"), val = tensor([1])]; + tensor x_73_cast_fp16 = expand_dims(axes = x_73_axes_0, x = V_layer_cache_9_cast_fp16)[name = string("x_73_cast_fp16")]; + tensor var_3304 = const()[name = string("op_3304"), val = tensor([1, 2, 1, 1])]; + tensor x_75_cast_fp16 = tile(reps = var_3304, x = x_73_cast_fp16)[name = string("x_75_cast_fp16")]; + bool var_3331_transpose_x_0 = const()[name = string("op_3331_transpose_x_0"), val = bool(false)]; + bool var_3331_transpose_y_0 = const()[name = string("op_3331_transpose_y_0"), val = bool(true)]; + tensor var_3331 = matmul(transpose_x = var_3331_transpose_x_0, transpose_y = var_3331_transpose_y_0, x = query_states_35, y = key_states_47_cast_fp16)[name = string("op_3331")]; + fp16 var_3332_to_fp16 = const()[name = string("op_3332_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_17_cast_fp16 = mul(x = var_3331, y = var_3332_to_fp16)[name = string("attn_weights_17_cast_fp16")]; + tensor attn_weights_19_cast_fp16 = add(x = attn_weights_17_cast_fp16, y = causal_mask)[name = string("attn_weights_19_cast_fp16")]; + int32 var_3367 = const()[name = string("op_3367"), val = int32(-1)]; + tensor var_3369_cast_fp16 = softmax(axis = var_3367, x = attn_weights_19_cast_fp16)[name = string("op_3369_cast_fp16")]; + tensor concat_84 = const()[name = string("concat_84"), val = tensor([16, 128, 1024])]; + tensor reshape_12_cast_fp16 = reshape(shape = concat_84, x = var_3369_cast_fp16)[name = string("reshape_12_cast_fp16")]; + tensor concat_85 = const()[name = string("concat_85"), val = tensor([16, 1024, 128])]; + tensor reshape_13_cast_fp16 = reshape(shape = concat_85, x = x_75_cast_fp16)[name = string("reshape_13_cast_fp16")]; + bool matmul_4_transpose_x_0 = const()[name = string("matmul_4_transpose_x_0"), val = bool(false)]; + bool matmul_4_transpose_y_0 = const()[name = string("matmul_4_transpose_y_0"), val = bool(false)]; + tensor matmul_4_cast_fp16 = matmul(transpose_x = matmul_4_transpose_x_0, transpose_y = matmul_4_transpose_y_0, x = reshape_12_cast_fp16, y = reshape_13_cast_fp16)[name = string("matmul_4_cast_fp16")]; + tensor concat_89 = const()[name = string("concat_89"), val = tensor([1, 16, 128, 128])]; + tensor reshape_14_cast_fp16 = reshape(shape = concat_89, x = matmul_4_cast_fp16)[name = string("reshape_14_cast_fp16")]; + tensor var_3381_perm_0 = const()[name = string("op_3381_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_3400 = const()[name = string("op_3400"), val = tensor([1, 128, 2048])]; + tensor var_3381_cast_fp16 = transpose(perm = var_3381_perm_0, x = reshape_14_cast_fp16)[name = string("transpose_85")]; + tensor attn_output_45_cast_fp16 = reshape(shape = var_3400, x = var_3381_cast_fp16)[name = string("attn_output_45_cast_fp16")]; + tensor var_3405 = const()[name = string("op_3405"), val = tensor([0, 2, 1])]; + string var_3421_pad_type_0 = const()[name = string("op_3421_pad_type_0"), val = string("valid")]; + int32 var_3421_groups_0 = const()[name = string("op_3421_groups_0"), val = int32(1)]; + tensor var_3421_strides_0 = const()[name = string("op_3421_strides_0"), val = tensor([1])]; + tensor var_3421_pad_0 = const()[name = string("op_3421_pad_0"), val = tensor([0, 0])]; + tensor var_3421_dilations_0 = const()[name = string("op_3421_dilations_0"), val = tensor([1])]; + tensor squeeze_4_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680840064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685034432))))[name = string("squeeze_4_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_3406_cast_fp16 = transpose(perm = var_3405, x = attn_output_45_cast_fp16)[name = string("transpose_84")]; + tensor var_3421_cast_fp16 = conv(dilations = var_3421_dilations_0, groups = var_3421_groups_0, pad = var_3421_pad_0, pad_type = var_3421_pad_type_0, strides = var_3421_strides_0, weight = squeeze_4_cast_fp16_to_fp32_to_fp16_palettized, x = var_3406_cast_fp16)[name = string("op_3421_cast_fp16")]; + tensor var_3425 = const()[name = string("op_3425"), val = tensor([0, 2, 1])]; + tensor attn_output_49_cast_fp16 = transpose(perm = var_3425, x = var_3421_cast_fp16)[name = string("transpose_83")]; + tensor hidden_states_49_cast_fp16 = add(x = hidden_states_41_cast_fp16, y = attn_output_49_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; + int32 var_3438 = const()[name = string("op_3438"), val = int32(-1)]; + fp16 const_167_promoted_to_fp16 = const()[name = string("const_167_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3440_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_167_promoted_to_fp16)[name = string("op_3440_cast_fp16")]; + bool input_83_interleave_0 = const()[name = string("input_83_interleave_0"), val = bool(false)]; + tensor input_83_cast_fp16 = concat(axis = var_3438, interleave = input_83_interleave_0, values = (hidden_states_49_cast_fp16, var_3440_cast_fp16))[name = string("input_83_cast_fp16")]; + tensor normed_77_axes_0 = const()[name = string("normed_77_axes_0"), val = tensor([-1])]; + fp16 var_3435_to_fp16 = const()[name = string("op_3435_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_77_cast_fp16 = layer_norm(axes = normed_77_axes_0, epsilon = var_3435_to_fp16, x = input_83_cast_fp16)[name = string("normed_77_cast_fp16")]; + tensor normed_79_begin_0 = const()[name = string("normed_79_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_79_end_0 = const()[name = string("normed_79_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_79_end_mask_0 = const()[name = string("normed_79_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_79_cast_fp16 = slice_by_index(begin = normed_79_begin_0, end = normed_79_end_0, end_mask = normed_79_end_mask_0, x = normed_77_cast_fp16)[name = string("normed_79_cast_fp16")]; + tensor const_170_promoted_to_fp16 = const()[name = string("const_170_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685165568)))]; + tensor x_77_cast_fp16 = mul(x = normed_79_cast_fp16, y = const_170_promoted_to_fp16)[name = string("x_77_cast_fp16")]; + tensor var_3465 = const()[name = string("op_3465"), val = tensor([0, 2, 1])]; + tensor input_85_axes_0 = const()[name = string("input_85_axes_0"), val = tensor([2])]; + tensor var_3466 = transpose(perm = var_3465, x = x_77_cast_fp16)[name = string("transpose_82")]; + tensor input_85 = expand_dims(axes = input_85_axes_0, x = var_3466)[name = string("input_85")]; + string input_87_pad_type_0 = const()[name = string("input_87_pad_type_0"), val = string("valid")]; + tensor input_87_strides_0 = const()[name = string("input_87_strides_0"), val = tensor([1, 1])]; + tensor input_87_pad_0 = const()[name = string("input_87_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_87_dilations_0 = const()[name = string("input_87_dilations_0"), val = tensor([1, 1])]; + int32 input_87_groups_0 = const()[name = string("input_87_groups_0"), val = int32(1)]; + tensor input_87 = conv(dilations = input_87_dilations_0, groups = input_87_groups_0, pad = input_87_pad_0, pad_type = input_87_pad_type_0, strides = input_87_strides_0, weight = model_model_layers_4_mlp_gate_proj_weight_palettized, x = input_85)[name = string("input_87")]; + string b_9_pad_type_0 = const()[name = string("b_9_pad_type_0"), val = string("valid")]; + tensor b_9_strides_0 = const()[name = string("b_9_strides_0"), val = tensor([1, 1])]; + tensor b_9_pad_0 = const()[name = string("b_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_9_dilations_0 = const()[name = string("b_9_dilations_0"), val = tensor([1, 1])]; + int32 b_9_groups_0 = const()[name = string("b_9_groups_0"), val = int32(1)]; + tensor b_9 = conv(dilations = b_9_dilations_0, groups = b_9_groups_0, pad = b_9_pad_0, pad_type = b_9_pad_type_0, strides = b_9_strides_0, weight = model_model_layers_4_mlp_up_proj_weight_palettized, x = input_85)[name = string("b_9")]; + tensor c_9 = silu(x = input_87)[name = string("c_9")]; + tensor input_89 = mul(x = c_9, y = b_9)[name = string("input_89")]; + string e_9_pad_type_0 = const()[name = string("e_9_pad_type_0"), val = string("valid")]; + tensor e_9_strides_0 = const()[name = string("e_9_strides_0"), val = tensor([1, 1])]; + tensor e_9_pad_0 = const()[name = string("e_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_9_dilations_0 = const()[name = string("e_9_dilations_0"), val = tensor([1, 1])]; + int32 e_9_groups_0 = const()[name = string("e_9_groups_0"), val = int32(1)]; + tensor e_9 = conv(dilations = e_9_dilations_0, groups = e_9_groups_0, pad = e_9_pad_0, pad_type = e_9_pad_type_0, strides = e_9_strides_0, weight = model_model_layers_4_mlp_down_proj_weight_palettized, x = input_89)[name = string("e_9")]; + tensor var_3488_axes_0 = const()[name = string("op_3488_axes_0"), val = tensor([2])]; + tensor var_3488 = squeeze(axes = var_3488_axes_0, x = e_9)[name = string("op_3488")]; + tensor var_3489 = const()[name = string("op_3489"), val = tensor([0, 2, 1])]; + tensor var_3490 = transpose(perm = var_3489, x = var_3488)[name = string("transpose_81")]; + tensor hidden_states_51_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = var_3490)[name = string("hidden_states_51_cast_fp16")]; + int32 var_3502 = const()[name = string("op_3502"), val = int32(-1)]; + fp16 const_171_promoted_to_fp16 = const()[name = string("const_171_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3504_cast_fp16 = mul(x = hidden_states_51_cast_fp16, y = const_171_promoted_to_fp16)[name = string("op_3504_cast_fp16")]; + bool input_91_interleave_0 = const()[name = string("input_91_interleave_0"), val = bool(false)]; + tensor input_91_cast_fp16 = concat(axis = var_3502, interleave = input_91_interleave_0, values = (hidden_states_51_cast_fp16, var_3504_cast_fp16))[name = string("input_91_cast_fp16")]; + tensor normed_81_axes_0 = const()[name = string("normed_81_axes_0"), val = tensor([-1])]; + fp16 var_3499_to_fp16 = const()[name = string("op_3499_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_81_cast_fp16 = layer_norm(axes = normed_81_axes_0, epsilon = var_3499_to_fp16, x = input_91_cast_fp16)[name = string("normed_81_cast_fp16")]; + tensor normed_83_begin_0 = const()[name = string("normed_83_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_83_end_0 = const()[name = string("normed_83_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_83_end_mask_0 = const()[name = string("normed_83_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_83_cast_fp16 = slice_by_index(begin = normed_83_begin_0, end = normed_83_end_0, end_mask = normed_83_end_mask_0, x = normed_81_cast_fp16)[name = string("normed_83_cast_fp16")]; + tensor const_174_promoted_to_fp16 = const()[name = string("const_174_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685169728)))]; + tensor hidden_states_53_cast_fp16 = mul(x = normed_83_cast_fp16, y = const_174_promoted_to_fp16)[name = string("hidden_states_53_cast_fp16")]; + tensor var_3527 = const()[name = string("op_3527"), val = tensor([0, 2, 1])]; + tensor var_3530_axes_0 = const()[name = string("op_3530_axes_0"), val = tensor([2])]; + tensor var_3528_cast_fp16 = transpose(perm = var_3527, x = hidden_states_53_cast_fp16)[name = string("transpose_80")]; + tensor var_3530_cast_fp16 = expand_dims(axes = var_3530_axes_0, x = var_3528_cast_fp16)[name = string("op_3530_cast_fp16")]; + string query_states_41_pad_type_0 = const()[name = string("query_states_41_pad_type_0"), val = string("valid")]; + tensor query_states_41_strides_0 = const()[name = string("query_states_41_strides_0"), val = tensor([1, 1])]; + tensor query_states_41_pad_0 = const()[name = string("query_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_41_dilations_0 = const()[name = string("query_states_41_dilations_0"), val = tensor([1, 1])]; + int32 query_states_41_groups_0 = const()[name = string("query_states_41_groups_0"), val = int32(1)]; + tensor query_states_41 = conv(dilations = query_states_41_dilations_0, groups = query_states_41_groups_0, pad = query_states_41_pad_0, pad_type = query_states_41_pad_type_0, strides = query_states_41_strides_0, weight = model_model_layers_5_self_attn_q_proj_weight_palettized, x = var_3530_cast_fp16)[name = string("query_states_41")]; + string key_states_51_pad_type_0 = const()[name = string("key_states_51_pad_type_0"), val = string("valid")]; + tensor key_states_51_strides_0 = const()[name = string("key_states_51_strides_0"), val = tensor([1, 1])]; + tensor key_states_51_pad_0 = const()[name = string("key_states_51_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_51_dilations_0 = const()[name = string("key_states_51_dilations_0"), val = tensor([1, 1])]; + int32 key_states_51_groups_0 = const()[name = string("key_states_51_groups_0"), val = int32(1)]; + tensor key_states_51 = conv(dilations = key_states_51_dilations_0, groups = key_states_51_groups_0, pad = key_states_51_pad_0, pad_type = key_states_51_pad_type_0, strides = key_states_51_strides_0, weight = model_model_layers_5_self_attn_k_proj_weight_palettized, x = var_3530_cast_fp16)[name = string("key_states_51")]; + string value_states_41_pad_type_0 = const()[name = string("value_states_41_pad_type_0"), val = string("valid")]; + tensor value_states_41_strides_0 = const()[name = string("value_states_41_strides_0"), val = tensor([1, 1])]; + tensor value_states_41_pad_0 = const()[name = string("value_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_41_dilations_0 = const()[name = string("value_states_41_dilations_0"), val = tensor([1, 1])]; + int32 value_states_41_groups_0 = const()[name = string("value_states_41_groups_0"), val = int32(1)]; + tensor value_states_41 = conv(dilations = value_states_41_dilations_0, groups = value_states_41_groups_0, pad = value_states_41_pad_0, pad_type = value_states_41_pad_type_0, strides = value_states_41_strides_0, weight = model_model_layers_5_self_attn_v_proj_weight_palettized, x = var_3530_cast_fp16)[name = string("value_states_41")]; + tensor var_3572 = const()[name = string("op_3572"), val = tensor([1, 16, 128, 128])]; + tensor var_3573 = reshape(shape = var_3572, x = query_states_41)[name = string("op_3573")]; + tensor var_3578 = const()[name = string("op_3578"), val = tensor([0, 1, 3, 2])]; + tensor var_3583 = const()[name = string("op_3583"), val = tensor([1, 8, 128, 128])]; + tensor var_3584 = reshape(shape = var_3583, x = key_states_51)[name = string("op_3584")]; + tensor var_3589 = const()[name = string("op_3589"), val = tensor([0, 1, 3, 2])]; + tensor var_3594 = const()[name = string("op_3594"), val = tensor([1, 8, 128, 128])]; + tensor var_3595 = reshape(shape = var_3594, x = value_states_41)[name = string("op_3595")]; + tensor var_3600 = const()[name = string("op_3600"), val = tensor([0, 1, 3, 2])]; + int32 var_3611 = const()[name = string("op_3611"), val = int32(-1)]; + fp16 const_176_promoted = const()[name = string("const_176_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_55 = transpose(perm = var_3578, x = var_3573)[name = string("transpose_79")]; + tensor var_3613 = mul(x = hidden_states_55, y = const_176_promoted)[name = string("op_3613")]; + bool input_95_interleave_0 = const()[name = string("input_95_interleave_0"), val = bool(false)]; + tensor input_95 = concat(axis = var_3611, interleave = input_95_interleave_0, values = (hidden_states_55, var_3613))[name = string("input_95")]; + tensor normed_85_axes_0 = const()[name = string("normed_85_axes_0"), val = tensor([-1])]; + fp16 var_3608_to_fp16 = const()[name = string("op_3608_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_85_cast_fp16 = layer_norm(axes = normed_85_axes_0, epsilon = var_3608_to_fp16, x = input_95)[name = string("normed_85_cast_fp16")]; + tensor normed_87_begin_0 = const()[name = string("normed_87_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_87_end_0 = const()[name = string("normed_87_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_87_end_mask_0 = const()[name = string("normed_87_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_87 = slice_by_index(begin = normed_87_begin_0, end = normed_87_end_0, end_mask = normed_87_end_mask_0, x = normed_85_cast_fp16)[name = string("normed_87")]; + tensor const_179 = const()[name = string("const_179"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685173888)))]; + tensor q_11 = mul(x = normed_87, y = const_179)[name = string("q_11")]; + int32 var_3636 = const()[name = string("op_3636"), val = int32(-1)]; + fp16 const_180_promoted = const()[name = string("const_180_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_57 = transpose(perm = var_3589, x = var_3584)[name = string("transpose_78")]; + tensor var_3638 = mul(x = hidden_states_57, y = const_180_promoted)[name = string("op_3638")]; + bool input_97_interleave_0 = const()[name = string("input_97_interleave_0"), val = bool(false)]; + tensor input_97 = concat(axis = var_3636, interleave = input_97_interleave_0, values = (hidden_states_57, var_3638))[name = string("input_97")]; + tensor normed_89_axes_0 = const()[name = string("normed_89_axes_0"), val = tensor([-1])]; + fp16 var_3633_to_fp16 = const()[name = string("op_3633_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_89_cast_fp16 = layer_norm(axes = normed_89_axes_0, epsilon = var_3633_to_fp16, x = input_97)[name = string("normed_89_cast_fp16")]; + tensor normed_91_begin_0 = const()[name = string("normed_91_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_91_end_0 = const()[name = string("normed_91_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_91_end_mask_0 = const()[name = string("normed_91_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_91 = slice_by_index(begin = normed_91_begin_0, end = normed_91_end_0, end_mask = normed_91_end_mask_0, x = normed_89_cast_fp16)[name = string("normed_91")]; + tensor const_183 = const()[name = string("const_183"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685174208)))]; + tensor k_11 = mul(x = normed_91, y = const_183)[name = string("k_11")]; + tensor var_3664 = mul(x = q_11, y = cos_5)[name = string("op_3664")]; + tensor x1_21_begin_0 = const()[name = string("x1_21_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_21_end_0 = const()[name = string("x1_21_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_21_end_mask_0 = const()[name = string("x1_21_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_21 = slice_by_index(begin = x1_21_begin_0, end = x1_21_end_0, end_mask = x1_21_end_mask_0, x = q_11)[name = string("x1_21")]; + tensor x2_21_begin_0 = const()[name = string("x2_21_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_21_end_0 = const()[name = string("x2_21_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_21_end_mask_0 = const()[name = string("x2_21_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_21 = slice_by_index(begin = x2_21_begin_0, end = x2_21_end_0, end_mask = x2_21_end_mask_0, x = q_11)[name = string("x2_21")]; + fp16 const_186_promoted = const()[name = string("const_186_promoted"), val = fp16(-0x1p+0)]; + tensor var_3685 = mul(x = x2_21, y = const_186_promoted)[name = string("op_3685")]; + int32 var_3687 = const()[name = string("op_3687"), val = int32(-1)]; + bool var_3688_interleave_0 = const()[name = string("op_3688_interleave_0"), val = bool(false)]; + tensor var_3688 = concat(axis = var_3687, interleave = var_3688_interleave_0, values = (var_3685, x1_21))[name = string("op_3688")]; + tensor var_3689 = mul(x = var_3688, y = sin_5)[name = string("op_3689")]; + tensor query_states_43 = add(x = var_3664, y = var_3689)[name = string("query_states_43")]; + tensor var_3692 = mul(x = k_11, y = cos_5)[name = string("op_3692")]; + tensor x1_23_begin_0 = const()[name = string("x1_23_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_23_end_0 = const()[name = string("x1_23_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_23_end_mask_0 = const()[name = string("x1_23_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_23 = slice_by_index(begin = x1_23_begin_0, end = x1_23_end_0, end_mask = x1_23_end_mask_0, x = k_11)[name = string("x1_23")]; + tensor x2_23_begin_0 = const()[name = string("x2_23_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_23_end_0 = const()[name = string("x2_23_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_23_end_mask_0 = const()[name = string("x2_23_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_23 = slice_by_index(begin = x2_23_begin_0, end = x2_23_end_0, end_mask = x2_23_end_mask_0, x = k_11)[name = string("x2_23")]; + fp16 const_189_promoted = const()[name = string("const_189_promoted"), val = fp16(-0x1p+0)]; + tensor var_3713 = mul(x = x2_23, y = const_189_promoted)[name = string("op_3713")]; + int32 var_3715 = const()[name = string("op_3715"), val = int32(-1)]; + bool var_3716_interleave_0 = const()[name = string("op_3716_interleave_0"), val = bool(false)]; + tensor var_3716 = concat(axis = var_3715, interleave = var_3716_interleave_0, values = (var_3713, x1_23))[name = string("op_3716")]; + tensor var_3717 = mul(x = var_3716, y = sin_5)[name = string("op_3717")]; + tensor key_states_53 = add(x = var_3692, y = var_3717)[name = string("key_states_53")]; + tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([5])]; + tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; + tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; + tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([6])]; + int32 concat_92_axis_0 = const()[name = string("concat_92_axis_0"), val = int32(0)]; + bool concat_92_interleave_0 = const()[name = string("concat_92_interleave_0"), val = bool(false)]; + tensor concat_92 = concat(axis = concat_92_axis_0, interleave = concat_92_interleave_0, values = (expand_dims_60, expand_dims_61, current_pos, expand_dims_63))[name = string("concat_92")]; + tensor concat_93_values1_0 = const()[name = string("concat_93_values1_0"), val = tensor([0])]; + tensor concat_93_values3_0 = const()[name = string("concat_93_values3_0"), val = tensor([0])]; + int32 concat_93_axis_0 = const()[name = string("concat_93_axis_0"), val = int32(0)]; + bool concat_93_interleave_0 = const()[name = string("concat_93_interleave_0"), val = bool(false)]; + tensor concat_93 = concat(axis = concat_93_axis_0, interleave = concat_93_interleave_0, values = (expand_dims_64, concat_93_values1_0, var_1039, concat_93_values3_0))[name = string("concat_93")]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_92, begin_mask = model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0, end = concat_93, end_mask = model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_11_stride_0, update = key_states_53, x = coreml_update_state_37)[name = string("model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_94_write_state")]; + tensor coreml_update_state_38 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_94")]; + tensor expand_dims_66 = const()[name = string("expand_dims_66"), val = tensor([33])]; + tensor expand_dims_67 = const()[name = string("expand_dims_67"), val = tensor([0])]; + tensor expand_dims_69 = const()[name = string("expand_dims_69"), val = tensor([0])]; + tensor expand_dims_70 = const()[name = string("expand_dims_70"), val = tensor([34])]; + int32 concat_96_axis_0 = const()[name = string("concat_96_axis_0"), val = int32(0)]; + bool concat_96_interleave_0 = const()[name = string("concat_96_interleave_0"), val = bool(false)]; + tensor concat_96 = concat(axis = concat_96_axis_0, interleave = concat_96_interleave_0, values = (expand_dims_66, expand_dims_67, current_pos, expand_dims_69))[name = string("concat_96")]; + tensor concat_97_values1_0 = const()[name = string("concat_97_values1_0"), val = tensor([0])]; + tensor concat_97_values3_0 = const()[name = string("concat_97_values3_0"), val = tensor([0])]; + int32 concat_97_axis_0 = const()[name = string("concat_97_axis_0"), val = int32(0)]; + bool concat_97_interleave_0 = const()[name = string("concat_97_interleave_0"), val = bool(false)]; + tensor concat_97 = concat(axis = concat_97_axis_0, interleave = concat_97_interleave_0, values = (expand_dims_70, concat_97_values1_0, var_1039, concat_97_values3_0))[name = string("concat_97")]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_43 = transpose(perm = var_3600, x = var_3595)[name = string("transpose_77")]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_96, begin_mask = model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0, end = concat_97, end_mask = model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_12_stride_0, update = value_states_43, x = coreml_update_state_38)[name = string("model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_95_write_state")]; + tensor coreml_update_state_39 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_95")]; + tensor var_3788_begin_0 = const()[name = string("op_3788_begin_0"), val = tensor([5, 0, 0, 0])]; + tensor var_3788_end_0 = const()[name = string("op_3788_end_0"), val = tensor([6, 8, 1024, 128])]; + tensor var_3788_end_mask_0 = const()[name = string("op_3788_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3788_cast_fp16 = slice_by_index(begin = var_3788_begin_0, end = var_3788_end_0, end_mask = var_3788_end_mask_0, x = coreml_update_state_39)[name = string("op_3788_cast_fp16")]; + tensor K_layer_cache_11_axes_0 = const()[name = string("K_layer_cache_11_axes_0"), val = tensor([0])]; + tensor K_layer_cache_11_cast_fp16 = squeeze(axes = K_layer_cache_11_axes_0, x = var_3788_cast_fp16)[name = string("K_layer_cache_11_cast_fp16")]; + tensor var_3795_begin_0 = const()[name = string("op_3795_begin_0"), val = tensor([33, 0, 0, 0])]; + tensor var_3795_end_0 = const()[name = string("op_3795_end_0"), val = tensor([34, 8, 1024, 128])]; + tensor var_3795_end_mask_0 = const()[name = string("op_3795_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3795_cast_fp16 = slice_by_index(begin = var_3795_begin_0, end = var_3795_end_0, end_mask = var_3795_end_mask_0, x = coreml_update_state_39)[name = string("op_3795_cast_fp16")]; + tensor V_layer_cache_11_axes_0 = const()[name = string("V_layer_cache_11_axes_0"), val = tensor([0])]; + tensor V_layer_cache_11_cast_fp16 = squeeze(axes = V_layer_cache_11_axes_0, x = var_3795_cast_fp16)[name = string("V_layer_cache_11_cast_fp16")]; + tensor x_83_axes_0 = const()[name = string("x_83_axes_0"), val = tensor([1])]; + tensor x_83_cast_fp16 = expand_dims(axes = x_83_axes_0, x = K_layer_cache_11_cast_fp16)[name = string("x_83_cast_fp16")]; + tensor var_3824 = const()[name = string("op_3824"), val = tensor([1, 2, 1, 1])]; + tensor x_85_cast_fp16 = tile(reps = var_3824, x = x_83_cast_fp16)[name = string("x_85_cast_fp16")]; + tensor var_3836 = const()[name = string("op_3836"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_57_cast_fp16 = reshape(shape = var_3836, x = x_85_cast_fp16)[name = string("key_states_57_cast_fp16")]; + tensor x_89_axes_0 = const()[name = string("x_89_axes_0"), val = tensor([1])]; + tensor x_89_cast_fp16 = expand_dims(axes = x_89_axes_0, x = V_layer_cache_11_cast_fp16)[name = string("x_89_cast_fp16")]; + tensor var_3844 = const()[name = string("op_3844"), val = tensor([1, 2, 1, 1])]; + tensor x_91_cast_fp16 = tile(reps = var_3844, x = x_89_cast_fp16)[name = string("x_91_cast_fp16")]; + bool var_3871_transpose_x_0 = const()[name = string("op_3871_transpose_x_0"), val = bool(false)]; + bool var_3871_transpose_y_0 = const()[name = string("op_3871_transpose_y_0"), val = bool(true)]; + tensor var_3871 = matmul(transpose_x = var_3871_transpose_x_0, transpose_y = var_3871_transpose_y_0, x = query_states_43, y = key_states_57_cast_fp16)[name = string("op_3871")]; + fp16 var_3872_to_fp16 = const()[name = string("op_3872_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_21_cast_fp16 = mul(x = var_3871, y = var_3872_to_fp16)[name = string("attn_weights_21_cast_fp16")]; + tensor attn_weights_23_cast_fp16 = add(x = attn_weights_21_cast_fp16, y = causal_mask)[name = string("attn_weights_23_cast_fp16")]; + int32 var_3907 = const()[name = string("op_3907"), val = int32(-1)]; + tensor var_3909_cast_fp16 = softmax(axis = var_3907, x = attn_weights_23_cast_fp16)[name = string("op_3909_cast_fp16")]; + tensor concat_102 = const()[name = string("concat_102"), val = tensor([16, 128, 1024])]; + tensor reshape_15_cast_fp16 = reshape(shape = concat_102, x = var_3909_cast_fp16)[name = string("reshape_15_cast_fp16")]; + tensor concat_103 = const()[name = string("concat_103"), val = tensor([16, 1024, 128])]; + tensor reshape_16_cast_fp16 = reshape(shape = concat_103, x = x_91_cast_fp16)[name = string("reshape_16_cast_fp16")]; + bool matmul_5_transpose_x_0 = const()[name = string("matmul_5_transpose_x_0"), val = bool(false)]; + bool matmul_5_transpose_y_0 = const()[name = string("matmul_5_transpose_y_0"), val = bool(false)]; + tensor matmul_5_cast_fp16 = matmul(transpose_x = matmul_5_transpose_x_0, transpose_y = matmul_5_transpose_y_0, x = reshape_15_cast_fp16, y = reshape_16_cast_fp16)[name = string("matmul_5_cast_fp16")]; + tensor concat_107 = const()[name = string("concat_107"), val = tensor([1, 16, 128, 128])]; + tensor reshape_17_cast_fp16 = reshape(shape = concat_107, x = matmul_5_cast_fp16)[name = string("reshape_17_cast_fp16")]; + tensor var_3921_perm_0 = const()[name = string("op_3921_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_3940 = const()[name = string("op_3940"), val = tensor([1, 128, 2048])]; + tensor var_3921_cast_fp16 = transpose(perm = var_3921_perm_0, x = reshape_17_cast_fp16)[name = string("transpose_76")]; + tensor attn_output_55_cast_fp16 = reshape(shape = var_3940, x = var_3921_cast_fp16)[name = string("attn_output_55_cast_fp16")]; + tensor var_3945 = const()[name = string("op_3945"), val = tensor([0, 2, 1])]; + string var_3961_pad_type_0 = const()[name = string("op_3961_pad_type_0"), val = string("valid")]; + int32 var_3961_groups_0 = const()[name = string("op_3961_groups_0"), val = int32(1)]; + tensor var_3961_strides_0 = const()[name = string("op_3961_strides_0"), val = tensor([1])]; + tensor var_3961_pad_0 = const()[name = string("op_3961_pad_0"), val = tensor([0, 0])]; + tensor var_3961_dilations_0 = const()[name = string("op_3961_dilations_0"), val = tensor([1])]; + tensor squeeze_5_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685174528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689368896))))[name = string("squeeze_5_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_3946_cast_fp16 = transpose(perm = var_3945, x = attn_output_55_cast_fp16)[name = string("transpose_75")]; + tensor var_3961_cast_fp16 = conv(dilations = var_3961_dilations_0, groups = var_3961_groups_0, pad = var_3961_pad_0, pad_type = var_3961_pad_type_0, strides = var_3961_strides_0, weight = squeeze_5_cast_fp16_to_fp32_to_fp16_palettized, x = var_3946_cast_fp16)[name = string("op_3961_cast_fp16")]; + tensor var_3965 = const()[name = string("op_3965"), val = tensor([0, 2, 1])]; + tensor attn_output_59_cast_fp16 = transpose(perm = var_3965, x = var_3961_cast_fp16)[name = string("transpose_74")]; + tensor hidden_states_59_cast_fp16 = add(x = hidden_states_51_cast_fp16, y = attn_output_59_cast_fp16)[name = string("hidden_states_59_cast_fp16")]; + int32 var_3978 = const()[name = string("op_3978"), val = int32(-1)]; + fp16 const_201_promoted_to_fp16 = const()[name = string("const_201_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3980_cast_fp16 = mul(x = hidden_states_59_cast_fp16, y = const_201_promoted_to_fp16)[name = string("op_3980_cast_fp16")]; + bool input_101_interleave_0 = const()[name = string("input_101_interleave_0"), val = bool(false)]; + tensor input_101_cast_fp16 = concat(axis = var_3978, interleave = input_101_interleave_0, values = (hidden_states_59_cast_fp16, var_3980_cast_fp16))[name = string("input_101_cast_fp16")]; + tensor normed_93_axes_0 = const()[name = string("normed_93_axes_0"), val = tensor([-1])]; + fp16 var_3975_to_fp16 = const()[name = string("op_3975_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_93_cast_fp16 = layer_norm(axes = normed_93_axes_0, epsilon = var_3975_to_fp16, x = input_101_cast_fp16)[name = string("normed_93_cast_fp16")]; + tensor normed_95_begin_0 = const()[name = string("normed_95_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_95_end_0 = const()[name = string("normed_95_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_95_end_mask_0 = const()[name = string("normed_95_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_95_cast_fp16 = slice_by_index(begin = normed_95_begin_0, end = normed_95_end_0, end_mask = normed_95_end_mask_0, x = normed_93_cast_fp16)[name = string("normed_95_cast_fp16")]; + tensor const_204_promoted_to_fp16 = const()[name = string("const_204_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689500032)))]; + tensor x_93_cast_fp16 = mul(x = normed_95_cast_fp16, y = const_204_promoted_to_fp16)[name = string("x_93_cast_fp16")]; + tensor var_4005 = const()[name = string("op_4005"), val = tensor([0, 2, 1])]; + tensor input_103_axes_0 = const()[name = string("input_103_axes_0"), val = tensor([2])]; + tensor var_4006 = transpose(perm = var_4005, x = x_93_cast_fp16)[name = string("transpose_73")]; + tensor input_103 = expand_dims(axes = input_103_axes_0, x = var_4006)[name = string("input_103")]; + string input_105_pad_type_0 = const()[name = string("input_105_pad_type_0"), val = string("valid")]; + tensor input_105_strides_0 = const()[name = string("input_105_strides_0"), val = tensor([1, 1])]; + tensor input_105_pad_0 = const()[name = string("input_105_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_105_dilations_0 = const()[name = string("input_105_dilations_0"), val = tensor([1, 1])]; + int32 input_105_groups_0 = const()[name = string("input_105_groups_0"), val = int32(1)]; + tensor input_105 = conv(dilations = input_105_dilations_0, groups = input_105_groups_0, pad = input_105_pad_0, pad_type = input_105_pad_type_0, strides = input_105_strides_0, weight = model_model_layers_5_mlp_gate_proj_weight_palettized, x = input_103)[name = string("input_105")]; + string b_11_pad_type_0 = const()[name = string("b_11_pad_type_0"), val = string("valid")]; + tensor b_11_strides_0 = const()[name = string("b_11_strides_0"), val = tensor([1, 1])]; + tensor b_11_pad_0 = const()[name = string("b_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_11_dilations_0 = const()[name = string("b_11_dilations_0"), val = tensor([1, 1])]; + int32 b_11_groups_0 = const()[name = string("b_11_groups_0"), val = int32(1)]; + tensor b_11 = conv(dilations = b_11_dilations_0, groups = b_11_groups_0, pad = b_11_pad_0, pad_type = b_11_pad_type_0, strides = b_11_strides_0, weight = model_model_layers_5_mlp_up_proj_weight_palettized, x = input_103)[name = string("b_11")]; + tensor c_11 = silu(x = input_105)[name = string("c_11")]; + tensor input_107 = mul(x = c_11, y = b_11)[name = string("input_107")]; + string e_11_pad_type_0 = const()[name = string("e_11_pad_type_0"), val = string("valid")]; + tensor e_11_strides_0 = const()[name = string("e_11_strides_0"), val = tensor([1, 1])]; + tensor e_11_pad_0 = const()[name = string("e_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_11_dilations_0 = const()[name = string("e_11_dilations_0"), val = tensor([1, 1])]; + int32 e_11_groups_0 = const()[name = string("e_11_groups_0"), val = int32(1)]; + tensor e_11 = conv(dilations = e_11_dilations_0, groups = e_11_groups_0, pad = e_11_pad_0, pad_type = e_11_pad_type_0, strides = e_11_strides_0, weight = model_model_layers_5_mlp_down_proj_weight_palettized, x = input_107)[name = string("e_11")]; + tensor var_4028_axes_0 = const()[name = string("op_4028_axes_0"), val = tensor([2])]; + tensor var_4028 = squeeze(axes = var_4028_axes_0, x = e_11)[name = string("op_4028")]; + tensor var_4029 = const()[name = string("op_4029"), val = tensor([0, 2, 1])]; + tensor var_4030 = transpose(perm = var_4029, x = var_4028)[name = string("transpose_72")]; + tensor hidden_states_61_cast_fp16 = add(x = hidden_states_59_cast_fp16, y = var_4030)[name = string("hidden_states_61_cast_fp16")]; + int32 var_4042 = const()[name = string("op_4042"), val = int32(-1)]; + fp16 const_205_promoted_to_fp16 = const()[name = string("const_205_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4044_cast_fp16 = mul(x = hidden_states_61_cast_fp16, y = const_205_promoted_to_fp16)[name = string("op_4044_cast_fp16")]; + bool input_109_interleave_0 = const()[name = string("input_109_interleave_0"), val = bool(false)]; + tensor input_109_cast_fp16 = concat(axis = var_4042, interleave = input_109_interleave_0, values = (hidden_states_61_cast_fp16, var_4044_cast_fp16))[name = string("input_109_cast_fp16")]; + tensor normed_97_axes_0 = const()[name = string("normed_97_axes_0"), val = tensor([-1])]; + fp16 var_4039_to_fp16 = const()[name = string("op_4039_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_97_cast_fp16 = layer_norm(axes = normed_97_axes_0, epsilon = var_4039_to_fp16, x = input_109_cast_fp16)[name = string("normed_97_cast_fp16")]; + tensor normed_99_begin_0 = const()[name = string("normed_99_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_99_end_0 = const()[name = string("normed_99_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_99_end_mask_0 = const()[name = string("normed_99_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_99_cast_fp16 = slice_by_index(begin = normed_99_begin_0, end = normed_99_end_0, end_mask = normed_99_end_mask_0, x = normed_97_cast_fp16)[name = string("normed_99_cast_fp16")]; + tensor const_208_promoted_to_fp16 = const()[name = string("const_208_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689504192)))]; + tensor hidden_states_63_cast_fp16 = mul(x = normed_99_cast_fp16, y = const_208_promoted_to_fp16)[name = string("hidden_states_63_cast_fp16")]; + tensor var_4067 = const()[name = string("op_4067"), val = tensor([0, 2, 1])]; + tensor var_4070_axes_0 = const()[name = string("op_4070_axes_0"), val = tensor([2])]; + tensor var_4068_cast_fp16 = transpose(perm = var_4067, x = hidden_states_63_cast_fp16)[name = string("transpose_71")]; + tensor var_4070_cast_fp16 = expand_dims(axes = var_4070_axes_0, x = var_4068_cast_fp16)[name = string("op_4070_cast_fp16")]; + string query_states_49_pad_type_0 = const()[name = string("query_states_49_pad_type_0"), val = string("valid")]; + tensor query_states_49_strides_0 = const()[name = string("query_states_49_strides_0"), val = tensor([1, 1])]; + tensor query_states_49_pad_0 = const()[name = string("query_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_49_dilations_0 = const()[name = string("query_states_49_dilations_0"), val = tensor([1, 1])]; + int32 query_states_49_groups_0 = const()[name = string("query_states_49_groups_0"), val = int32(1)]; + tensor query_states_49 = conv(dilations = query_states_49_dilations_0, groups = query_states_49_groups_0, pad = query_states_49_pad_0, pad_type = query_states_49_pad_type_0, strides = query_states_49_strides_0, weight = model_model_layers_6_self_attn_q_proj_weight_palettized, x = var_4070_cast_fp16)[name = string("query_states_49")]; + string key_states_61_pad_type_0 = const()[name = string("key_states_61_pad_type_0"), val = string("valid")]; + tensor key_states_61_strides_0 = const()[name = string("key_states_61_strides_0"), val = tensor([1, 1])]; + tensor key_states_61_pad_0 = const()[name = string("key_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_61_dilations_0 = const()[name = string("key_states_61_dilations_0"), val = tensor([1, 1])]; + int32 key_states_61_groups_0 = const()[name = string("key_states_61_groups_0"), val = int32(1)]; + tensor key_states_61 = conv(dilations = key_states_61_dilations_0, groups = key_states_61_groups_0, pad = key_states_61_pad_0, pad_type = key_states_61_pad_type_0, strides = key_states_61_strides_0, weight = model_model_layers_6_self_attn_k_proj_weight_palettized, x = var_4070_cast_fp16)[name = string("key_states_61")]; + string value_states_49_pad_type_0 = const()[name = string("value_states_49_pad_type_0"), val = string("valid")]; + tensor value_states_49_strides_0 = const()[name = string("value_states_49_strides_0"), val = tensor([1, 1])]; + tensor value_states_49_pad_0 = const()[name = string("value_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_49_dilations_0 = const()[name = string("value_states_49_dilations_0"), val = tensor([1, 1])]; + int32 value_states_49_groups_0 = const()[name = string("value_states_49_groups_0"), val = int32(1)]; + tensor value_states_49 = conv(dilations = value_states_49_dilations_0, groups = value_states_49_groups_0, pad = value_states_49_pad_0, pad_type = value_states_49_pad_type_0, strides = value_states_49_strides_0, weight = model_model_layers_6_self_attn_v_proj_weight_palettized, x = var_4070_cast_fp16)[name = string("value_states_49")]; + tensor var_4112 = const()[name = string("op_4112"), val = tensor([1, 16, 128, 128])]; + tensor var_4113 = reshape(shape = var_4112, x = query_states_49)[name = string("op_4113")]; + tensor var_4118 = const()[name = string("op_4118"), val = tensor([0, 1, 3, 2])]; + tensor var_4123 = const()[name = string("op_4123"), val = tensor([1, 8, 128, 128])]; + tensor var_4124 = reshape(shape = var_4123, x = key_states_61)[name = string("op_4124")]; + tensor var_4129 = const()[name = string("op_4129"), val = tensor([0, 1, 3, 2])]; + tensor var_4134 = const()[name = string("op_4134"), val = tensor([1, 8, 128, 128])]; + tensor var_4135 = reshape(shape = var_4134, x = value_states_49)[name = string("op_4135")]; + tensor var_4140 = const()[name = string("op_4140"), val = tensor([0, 1, 3, 2])]; + int32 var_4151 = const()[name = string("op_4151"), val = int32(-1)]; + fp16 const_210_promoted = const()[name = string("const_210_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_65 = transpose(perm = var_4118, x = var_4113)[name = string("transpose_70")]; + tensor var_4153 = mul(x = hidden_states_65, y = const_210_promoted)[name = string("op_4153")]; + bool input_113_interleave_0 = const()[name = string("input_113_interleave_0"), val = bool(false)]; + tensor input_113 = concat(axis = var_4151, interleave = input_113_interleave_0, values = (hidden_states_65, var_4153))[name = string("input_113")]; + tensor normed_101_axes_0 = const()[name = string("normed_101_axes_0"), val = tensor([-1])]; + fp16 var_4148_to_fp16 = const()[name = string("op_4148_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_101_cast_fp16 = layer_norm(axes = normed_101_axes_0, epsilon = var_4148_to_fp16, x = input_113)[name = string("normed_101_cast_fp16")]; + tensor normed_103_begin_0 = const()[name = string("normed_103_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_103_end_0 = const()[name = string("normed_103_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_103_end_mask_0 = const()[name = string("normed_103_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_103 = slice_by_index(begin = normed_103_begin_0, end = normed_103_end_0, end_mask = normed_103_end_mask_0, x = normed_101_cast_fp16)[name = string("normed_103")]; + tensor const_213 = const()[name = string("const_213"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689508352)))]; + tensor q_13 = mul(x = normed_103, y = const_213)[name = string("q_13")]; + int32 var_4176 = const()[name = string("op_4176"), val = int32(-1)]; + fp16 const_214_promoted = const()[name = string("const_214_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_67 = transpose(perm = var_4129, x = var_4124)[name = string("transpose_69")]; + tensor var_4178 = mul(x = hidden_states_67, y = const_214_promoted)[name = string("op_4178")]; + bool input_115_interleave_0 = const()[name = string("input_115_interleave_0"), val = bool(false)]; + tensor input_115 = concat(axis = var_4176, interleave = input_115_interleave_0, values = (hidden_states_67, var_4178))[name = string("input_115")]; + tensor normed_105_axes_0 = const()[name = string("normed_105_axes_0"), val = tensor([-1])]; + fp16 var_4173_to_fp16 = const()[name = string("op_4173_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_105_cast_fp16 = layer_norm(axes = normed_105_axes_0, epsilon = var_4173_to_fp16, x = input_115)[name = string("normed_105_cast_fp16")]; + tensor normed_107_begin_0 = const()[name = string("normed_107_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_107_end_0 = const()[name = string("normed_107_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_107_end_mask_0 = const()[name = string("normed_107_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_107 = slice_by_index(begin = normed_107_begin_0, end = normed_107_end_0, end_mask = normed_107_end_mask_0, x = normed_105_cast_fp16)[name = string("normed_107")]; + tensor const_217 = const()[name = string("const_217"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689508672)))]; + tensor k_13 = mul(x = normed_107, y = const_217)[name = string("k_13")]; + tensor var_4204 = mul(x = q_13, y = cos_5)[name = string("op_4204")]; + tensor x1_25_begin_0 = const()[name = string("x1_25_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_25_end_0 = const()[name = string("x1_25_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_25_end_mask_0 = const()[name = string("x1_25_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_25 = slice_by_index(begin = x1_25_begin_0, end = x1_25_end_0, end_mask = x1_25_end_mask_0, x = q_13)[name = string("x1_25")]; + tensor x2_25_begin_0 = const()[name = string("x2_25_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_25_end_0 = const()[name = string("x2_25_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_25_end_mask_0 = const()[name = string("x2_25_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_25 = slice_by_index(begin = x2_25_begin_0, end = x2_25_end_0, end_mask = x2_25_end_mask_0, x = q_13)[name = string("x2_25")]; + fp16 const_220_promoted = const()[name = string("const_220_promoted"), val = fp16(-0x1p+0)]; + tensor var_4225 = mul(x = x2_25, y = const_220_promoted)[name = string("op_4225")]; + int32 var_4227 = const()[name = string("op_4227"), val = int32(-1)]; + bool var_4228_interleave_0 = const()[name = string("op_4228_interleave_0"), val = bool(false)]; + tensor var_4228 = concat(axis = var_4227, interleave = var_4228_interleave_0, values = (var_4225, x1_25))[name = string("op_4228")]; + tensor var_4229 = mul(x = var_4228, y = sin_5)[name = string("op_4229")]; + tensor query_states_51 = add(x = var_4204, y = var_4229)[name = string("query_states_51")]; + tensor var_4232 = mul(x = k_13, y = cos_5)[name = string("op_4232")]; + tensor x1_27_begin_0 = const()[name = string("x1_27_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_27_end_0 = const()[name = string("x1_27_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_27_end_mask_0 = const()[name = string("x1_27_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_27 = slice_by_index(begin = x1_27_begin_0, end = x1_27_end_0, end_mask = x1_27_end_mask_0, x = k_13)[name = string("x1_27")]; + tensor x2_27_begin_0 = const()[name = string("x2_27_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_27_end_0 = const()[name = string("x2_27_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_27_end_mask_0 = const()[name = string("x2_27_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_27 = slice_by_index(begin = x2_27_begin_0, end = x2_27_end_0, end_mask = x2_27_end_mask_0, x = k_13)[name = string("x2_27")]; + fp16 const_223_promoted = const()[name = string("const_223_promoted"), val = fp16(-0x1p+0)]; + tensor var_4253 = mul(x = x2_27, y = const_223_promoted)[name = string("op_4253")]; + int32 var_4255 = const()[name = string("op_4255"), val = int32(-1)]; + bool var_4256_interleave_0 = const()[name = string("op_4256_interleave_0"), val = bool(false)]; + tensor var_4256 = concat(axis = var_4255, interleave = var_4256_interleave_0, values = (var_4253, x1_27))[name = string("op_4256")]; + tensor var_4257 = mul(x = var_4256, y = sin_5)[name = string("op_4257")]; + tensor key_states_63 = add(x = var_4232, y = var_4257)[name = string("key_states_63")]; + tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([6])]; + tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; + tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; + tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([7])]; + int32 concat_110_axis_0 = const()[name = string("concat_110_axis_0"), val = int32(0)]; + bool concat_110_interleave_0 = const()[name = string("concat_110_interleave_0"), val = bool(false)]; + tensor concat_110 = concat(axis = concat_110_axis_0, interleave = concat_110_interleave_0, values = (expand_dims_72, expand_dims_73, current_pos, expand_dims_75))[name = string("concat_110")]; + tensor concat_111_values1_0 = const()[name = string("concat_111_values1_0"), val = tensor([0])]; + tensor concat_111_values3_0 = const()[name = string("concat_111_values3_0"), val = tensor([0])]; + int32 concat_111_axis_0 = const()[name = string("concat_111_axis_0"), val = int32(0)]; + bool concat_111_interleave_0 = const()[name = string("concat_111_interleave_0"), val = bool(false)]; + tensor concat_111 = concat(axis = concat_111_axis_0, interleave = concat_111_interleave_0, values = (expand_dims_76, concat_111_values1_0, var_1039, concat_111_values3_0))[name = string("concat_111")]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_110, begin_mask = model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0, end = concat_111, end_mask = model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_13_stride_0, update = key_states_63, x = coreml_update_state_39)[name = string("model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_96_write_state")]; + tensor coreml_update_state_40 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_96")]; + tensor expand_dims_78 = const()[name = string("expand_dims_78"), val = tensor([34])]; + tensor expand_dims_79 = const()[name = string("expand_dims_79"), val = tensor([0])]; + tensor expand_dims_81 = const()[name = string("expand_dims_81"), val = tensor([0])]; + tensor expand_dims_82 = const()[name = string("expand_dims_82"), val = tensor([35])]; + int32 concat_114_axis_0 = const()[name = string("concat_114_axis_0"), val = int32(0)]; + bool concat_114_interleave_0 = const()[name = string("concat_114_interleave_0"), val = bool(false)]; + tensor concat_114 = concat(axis = concat_114_axis_0, interleave = concat_114_interleave_0, values = (expand_dims_78, expand_dims_79, current_pos, expand_dims_81))[name = string("concat_114")]; + tensor concat_115_values1_0 = const()[name = string("concat_115_values1_0"), val = tensor([0])]; + tensor concat_115_values3_0 = const()[name = string("concat_115_values3_0"), val = tensor([0])]; + int32 concat_115_axis_0 = const()[name = string("concat_115_axis_0"), val = int32(0)]; + bool concat_115_interleave_0 = const()[name = string("concat_115_interleave_0"), val = bool(false)]; + tensor concat_115 = concat(axis = concat_115_axis_0, interleave = concat_115_interleave_0, values = (expand_dims_82, concat_115_values1_0, var_1039, concat_115_values3_0))[name = string("concat_115")]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_51 = transpose(perm = var_4140, x = var_4135)[name = string("transpose_68")]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_114, begin_mask = model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0, end = concat_115, end_mask = model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_14_stride_0, update = value_states_51, x = coreml_update_state_40)[name = string("model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_97_write_state")]; + tensor coreml_update_state_41 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_97")]; + tensor var_4328_begin_0 = const()[name = string("op_4328_begin_0"), val = tensor([6, 0, 0, 0])]; + tensor var_4328_end_0 = const()[name = string("op_4328_end_0"), val = tensor([7, 8, 1024, 128])]; + tensor var_4328_end_mask_0 = const()[name = string("op_4328_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4328_cast_fp16 = slice_by_index(begin = var_4328_begin_0, end = var_4328_end_0, end_mask = var_4328_end_mask_0, x = coreml_update_state_41)[name = string("op_4328_cast_fp16")]; + tensor K_layer_cache_13_axes_0 = const()[name = string("K_layer_cache_13_axes_0"), val = tensor([0])]; + tensor K_layer_cache_13_cast_fp16 = squeeze(axes = K_layer_cache_13_axes_0, x = var_4328_cast_fp16)[name = string("K_layer_cache_13_cast_fp16")]; + tensor var_4335_begin_0 = const()[name = string("op_4335_begin_0"), val = tensor([34, 0, 0, 0])]; + tensor var_4335_end_0 = const()[name = string("op_4335_end_0"), val = tensor([35, 8, 1024, 128])]; + tensor var_4335_end_mask_0 = const()[name = string("op_4335_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4335_cast_fp16 = slice_by_index(begin = var_4335_begin_0, end = var_4335_end_0, end_mask = var_4335_end_mask_0, x = coreml_update_state_41)[name = string("op_4335_cast_fp16")]; + tensor V_layer_cache_13_axes_0 = const()[name = string("V_layer_cache_13_axes_0"), val = tensor([0])]; + tensor V_layer_cache_13_cast_fp16 = squeeze(axes = V_layer_cache_13_axes_0, x = var_4335_cast_fp16)[name = string("V_layer_cache_13_cast_fp16")]; + tensor x_99_axes_0 = const()[name = string("x_99_axes_0"), val = tensor([1])]; + tensor x_99_cast_fp16 = expand_dims(axes = x_99_axes_0, x = K_layer_cache_13_cast_fp16)[name = string("x_99_cast_fp16")]; + tensor var_4364 = const()[name = string("op_4364"), val = tensor([1, 2, 1, 1])]; + tensor x_101_cast_fp16 = tile(reps = var_4364, x = x_99_cast_fp16)[name = string("x_101_cast_fp16")]; + tensor var_4376 = const()[name = string("op_4376"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_67_cast_fp16 = reshape(shape = var_4376, x = x_101_cast_fp16)[name = string("key_states_67_cast_fp16")]; + tensor x_105_axes_0 = const()[name = string("x_105_axes_0"), val = tensor([1])]; + tensor x_105_cast_fp16 = expand_dims(axes = x_105_axes_0, x = V_layer_cache_13_cast_fp16)[name = string("x_105_cast_fp16")]; + tensor var_4384 = const()[name = string("op_4384"), val = tensor([1, 2, 1, 1])]; + tensor x_107_cast_fp16 = tile(reps = var_4384, x = x_105_cast_fp16)[name = string("x_107_cast_fp16")]; + bool var_4411_transpose_x_0 = const()[name = string("op_4411_transpose_x_0"), val = bool(false)]; + bool var_4411_transpose_y_0 = const()[name = string("op_4411_transpose_y_0"), val = bool(true)]; + tensor var_4411 = matmul(transpose_x = var_4411_transpose_x_0, transpose_y = var_4411_transpose_y_0, x = query_states_51, y = key_states_67_cast_fp16)[name = string("op_4411")]; + fp16 var_4412_to_fp16 = const()[name = string("op_4412_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_25_cast_fp16 = mul(x = var_4411, y = var_4412_to_fp16)[name = string("attn_weights_25_cast_fp16")]; + tensor attn_weights_27_cast_fp16 = add(x = attn_weights_25_cast_fp16, y = causal_mask)[name = string("attn_weights_27_cast_fp16")]; + int32 var_4447 = const()[name = string("op_4447"), val = int32(-1)]; + tensor var_4449_cast_fp16 = softmax(axis = var_4447, x = attn_weights_27_cast_fp16)[name = string("op_4449_cast_fp16")]; + tensor concat_120 = const()[name = string("concat_120"), val = tensor([16, 128, 1024])]; + tensor reshape_18_cast_fp16 = reshape(shape = concat_120, x = var_4449_cast_fp16)[name = string("reshape_18_cast_fp16")]; + tensor concat_121 = const()[name = string("concat_121"), val = tensor([16, 1024, 128])]; + tensor reshape_19_cast_fp16 = reshape(shape = concat_121, x = x_107_cast_fp16)[name = string("reshape_19_cast_fp16")]; + bool matmul_6_transpose_x_0 = const()[name = string("matmul_6_transpose_x_0"), val = bool(false)]; + bool matmul_6_transpose_y_0 = const()[name = string("matmul_6_transpose_y_0"), val = bool(false)]; + tensor matmul_6_cast_fp16 = matmul(transpose_x = matmul_6_transpose_x_0, transpose_y = matmul_6_transpose_y_0, x = reshape_18_cast_fp16, y = reshape_19_cast_fp16)[name = string("matmul_6_cast_fp16")]; + tensor concat_125 = const()[name = string("concat_125"), val = tensor([1, 16, 128, 128])]; + tensor reshape_20_cast_fp16 = reshape(shape = concat_125, x = matmul_6_cast_fp16)[name = string("reshape_20_cast_fp16")]; + tensor var_4461_perm_0 = const()[name = string("op_4461_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_4480 = const()[name = string("op_4480"), val = tensor([1, 128, 2048])]; + tensor var_4461_cast_fp16 = transpose(perm = var_4461_perm_0, x = reshape_20_cast_fp16)[name = string("transpose_67")]; + tensor attn_output_65_cast_fp16 = reshape(shape = var_4480, x = var_4461_cast_fp16)[name = string("attn_output_65_cast_fp16")]; + tensor var_4485 = const()[name = string("op_4485"), val = tensor([0, 2, 1])]; + string var_4501_pad_type_0 = const()[name = string("op_4501_pad_type_0"), val = string("valid")]; + int32 var_4501_groups_0 = const()[name = string("op_4501_groups_0"), val = int32(1)]; + tensor var_4501_strides_0 = const()[name = string("op_4501_strides_0"), val = tensor([1])]; + tensor var_4501_pad_0 = const()[name = string("op_4501_pad_0"), val = tensor([0, 0])]; + tensor var_4501_dilations_0 = const()[name = string("op_4501_dilations_0"), val = tensor([1])]; + tensor squeeze_6_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689508992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693703360))))[name = string("squeeze_6_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_4486_cast_fp16 = transpose(perm = var_4485, x = attn_output_65_cast_fp16)[name = string("transpose_66")]; + tensor var_4501_cast_fp16 = conv(dilations = var_4501_dilations_0, groups = var_4501_groups_0, pad = var_4501_pad_0, pad_type = var_4501_pad_type_0, strides = var_4501_strides_0, weight = squeeze_6_cast_fp16_to_fp32_to_fp16_palettized, x = var_4486_cast_fp16)[name = string("op_4501_cast_fp16")]; + tensor var_4505 = const()[name = string("op_4505"), val = tensor([0, 2, 1])]; + tensor attn_output_69_cast_fp16 = transpose(perm = var_4505, x = var_4501_cast_fp16)[name = string("transpose_65")]; + tensor hidden_states_69_cast_fp16 = add(x = hidden_states_61_cast_fp16, y = attn_output_69_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; + int32 var_4518 = const()[name = string("op_4518"), val = int32(-1)]; + fp16 const_235_promoted_to_fp16 = const()[name = string("const_235_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4520_cast_fp16 = mul(x = hidden_states_69_cast_fp16, y = const_235_promoted_to_fp16)[name = string("op_4520_cast_fp16")]; + bool input_119_interleave_0 = const()[name = string("input_119_interleave_0"), val = bool(false)]; + tensor input_119_cast_fp16 = concat(axis = var_4518, interleave = input_119_interleave_0, values = (hidden_states_69_cast_fp16, var_4520_cast_fp16))[name = string("input_119_cast_fp16")]; + tensor normed_109_axes_0 = const()[name = string("normed_109_axes_0"), val = tensor([-1])]; + fp16 var_4515_to_fp16 = const()[name = string("op_4515_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_109_cast_fp16 = layer_norm(axes = normed_109_axes_0, epsilon = var_4515_to_fp16, x = input_119_cast_fp16)[name = string("normed_109_cast_fp16")]; + tensor normed_111_begin_0 = const()[name = string("normed_111_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_111_end_0 = const()[name = string("normed_111_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_111_end_mask_0 = const()[name = string("normed_111_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_111_cast_fp16 = slice_by_index(begin = normed_111_begin_0, end = normed_111_end_0, end_mask = normed_111_end_mask_0, x = normed_109_cast_fp16)[name = string("normed_111_cast_fp16")]; + tensor const_238_promoted_to_fp16 = const()[name = string("const_238_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693834496)))]; + tensor x_109_cast_fp16 = mul(x = normed_111_cast_fp16, y = const_238_promoted_to_fp16)[name = string("x_109_cast_fp16")]; + tensor var_4545 = const()[name = string("op_4545"), val = tensor([0, 2, 1])]; + tensor input_121_axes_0 = const()[name = string("input_121_axes_0"), val = tensor([2])]; + tensor var_4546 = transpose(perm = var_4545, x = x_109_cast_fp16)[name = string("transpose_64")]; + tensor input_121 = expand_dims(axes = input_121_axes_0, x = var_4546)[name = string("input_121")]; + string input_123_pad_type_0 = const()[name = string("input_123_pad_type_0"), val = string("valid")]; + tensor input_123_strides_0 = const()[name = string("input_123_strides_0"), val = tensor([1, 1])]; + tensor input_123_pad_0 = const()[name = string("input_123_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_123_dilations_0 = const()[name = string("input_123_dilations_0"), val = tensor([1, 1])]; + int32 input_123_groups_0 = const()[name = string("input_123_groups_0"), val = int32(1)]; + tensor input_123 = conv(dilations = input_123_dilations_0, groups = input_123_groups_0, pad = input_123_pad_0, pad_type = input_123_pad_type_0, strides = input_123_strides_0, weight = model_model_layers_6_mlp_gate_proj_weight_palettized, x = input_121)[name = string("input_123")]; + string b_13_pad_type_0 = const()[name = string("b_13_pad_type_0"), val = string("valid")]; + tensor b_13_strides_0 = const()[name = string("b_13_strides_0"), val = tensor([1, 1])]; + tensor b_13_pad_0 = const()[name = string("b_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_13_dilations_0 = const()[name = string("b_13_dilations_0"), val = tensor([1, 1])]; + int32 b_13_groups_0 = const()[name = string("b_13_groups_0"), val = int32(1)]; + tensor b_13 = conv(dilations = b_13_dilations_0, groups = b_13_groups_0, pad = b_13_pad_0, pad_type = b_13_pad_type_0, strides = b_13_strides_0, weight = model_model_layers_6_mlp_up_proj_weight_palettized, x = input_121)[name = string("b_13")]; + tensor c_13 = silu(x = input_123)[name = string("c_13")]; + tensor input_125 = mul(x = c_13, y = b_13)[name = string("input_125")]; + string e_13_pad_type_0 = const()[name = string("e_13_pad_type_0"), val = string("valid")]; + tensor e_13_strides_0 = const()[name = string("e_13_strides_0"), val = tensor([1, 1])]; + tensor e_13_pad_0 = const()[name = string("e_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_13_dilations_0 = const()[name = string("e_13_dilations_0"), val = tensor([1, 1])]; + int32 e_13_groups_0 = const()[name = string("e_13_groups_0"), val = int32(1)]; + tensor e_13 = conv(dilations = e_13_dilations_0, groups = e_13_groups_0, pad = e_13_pad_0, pad_type = e_13_pad_type_0, strides = e_13_strides_0, weight = model_model_layers_6_mlp_down_proj_weight_palettized, x = input_125)[name = string("e_13")]; + tensor var_4568_axes_0 = const()[name = string("op_4568_axes_0"), val = tensor([2])]; + tensor var_4568 = squeeze(axes = var_4568_axes_0, x = e_13)[name = string("op_4568")]; + tensor var_4569 = const()[name = string("op_4569"), val = tensor([0, 2, 1])]; + tensor var_4570 = transpose(perm = var_4569, x = var_4568)[name = string("transpose_63")]; + tensor hidden_states_71_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = var_4570)[name = string("hidden_states_71_cast_fp16")]; + int32 var_4582 = const()[name = string("op_4582"), val = int32(-1)]; + fp16 const_239_promoted_to_fp16 = const()[name = string("const_239_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4584_cast_fp16 = mul(x = hidden_states_71_cast_fp16, y = const_239_promoted_to_fp16)[name = string("op_4584_cast_fp16")]; + bool input_127_interleave_0 = const()[name = string("input_127_interleave_0"), val = bool(false)]; + tensor input_127_cast_fp16 = concat(axis = var_4582, interleave = input_127_interleave_0, values = (hidden_states_71_cast_fp16, var_4584_cast_fp16))[name = string("input_127_cast_fp16")]; + tensor normed_113_axes_0 = const()[name = string("normed_113_axes_0"), val = tensor([-1])]; + fp16 var_4579_to_fp16 = const()[name = string("op_4579_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_113_cast_fp16 = layer_norm(axes = normed_113_axes_0, epsilon = var_4579_to_fp16, x = input_127_cast_fp16)[name = string("normed_113_cast_fp16")]; + tensor normed_115_begin_0 = const()[name = string("normed_115_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_115_end_0 = const()[name = string("normed_115_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_115_end_mask_0 = const()[name = string("normed_115_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_115_cast_fp16 = slice_by_index(begin = normed_115_begin_0, end = normed_115_end_0, end_mask = normed_115_end_mask_0, x = normed_113_cast_fp16)[name = string("normed_115_cast_fp16")]; + tensor const_242_promoted_to_fp16 = const()[name = string("const_242_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693838656)))]; + tensor hidden_states_73_cast_fp16 = mul(x = normed_115_cast_fp16, y = const_242_promoted_to_fp16)[name = string("hidden_states_73_cast_fp16")]; + tensor var_4607 = const()[name = string("op_4607"), val = tensor([0, 2, 1])]; + tensor var_4610_axes_0 = const()[name = string("op_4610_axes_0"), val = tensor([2])]; + tensor var_4608_cast_fp16 = transpose(perm = var_4607, x = hidden_states_73_cast_fp16)[name = string("transpose_62")]; + tensor var_4610_cast_fp16 = expand_dims(axes = var_4610_axes_0, x = var_4608_cast_fp16)[name = string("op_4610_cast_fp16")]; + string query_states_57_pad_type_0 = const()[name = string("query_states_57_pad_type_0"), val = string("valid")]; + tensor query_states_57_strides_0 = const()[name = string("query_states_57_strides_0"), val = tensor([1, 1])]; + tensor query_states_57_pad_0 = const()[name = string("query_states_57_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_57_dilations_0 = const()[name = string("query_states_57_dilations_0"), val = tensor([1, 1])]; + int32 query_states_57_groups_0 = const()[name = string("query_states_57_groups_0"), val = int32(1)]; + tensor query_states_57 = conv(dilations = query_states_57_dilations_0, groups = query_states_57_groups_0, pad = query_states_57_pad_0, pad_type = query_states_57_pad_type_0, strides = query_states_57_strides_0, weight = model_model_layers_7_self_attn_q_proj_weight_palettized, x = var_4610_cast_fp16)[name = string("query_states_57")]; + string key_states_71_pad_type_0 = const()[name = string("key_states_71_pad_type_0"), val = string("valid")]; + tensor key_states_71_strides_0 = const()[name = string("key_states_71_strides_0"), val = tensor([1, 1])]; + tensor key_states_71_pad_0 = const()[name = string("key_states_71_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_71_dilations_0 = const()[name = string("key_states_71_dilations_0"), val = tensor([1, 1])]; + int32 key_states_71_groups_0 = const()[name = string("key_states_71_groups_0"), val = int32(1)]; + tensor key_states_71 = conv(dilations = key_states_71_dilations_0, groups = key_states_71_groups_0, pad = key_states_71_pad_0, pad_type = key_states_71_pad_type_0, strides = key_states_71_strides_0, weight = model_model_layers_7_self_attn_k_proj_weight_palettized, x = var_4610_cast_fp16)[name = string("key_states_71")]; + string value_states_57_pad_type_0 = const()[name = string("value_states_57_pad_type_0"), val = string("valid")]; + tensor value_states_57_strides_0 = const()[name = string("value_states_57_strides_0"), val = tensor([1, 1])]; + tensor value_states_57_pad_0 = const()[name = string("value_states_57_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_57_dilations_0 = const()[name = string("value_states_57_dilations_0"), val = tensor([1, 1])]; + int32 value_states_57_groups_0 = const()[name = string("value_states_57_groups_0"), val = int32(1)]; + tensor value_states_57 = conv(dilations = value_states_57_dilations_0, groups = value_states_57_groups_0, pad = value_states_57_pad_0, pad_type = value_states_57_pad_type_0, strides = value_states_57_strides_0, weight = model_model_layers_7_self_attn_v_proj_weight_palettized, x = var_4610_cast_fp16)[name = string("value_states_57")]; + tensor var_4652 = const()[name = string("op_4652"), val = tensor([1, 16, 128, 128])]; + tensor var_4653 = reshape(shape = var_4652, x = query_states_57)[name = string("op_4653")]; + tensor var_4658 = const()[name = string("op_4658"), val = tensor([0, 1, 3, 2])]; + tensor var_4663 = const()[name = string("op_4663"), val = tensor([1, 8, 128, 128])]; + tensor var_4664 = reshape(shape = var_4663, x = key_states_71)[name = string("op_4664")]; + tensor var_4669 = const()[name = string("op_4669"), val = tensor([0, 1, 3, 2])]; + tensor var_4674 = const()[name = string("op_4674"), val = tensor([1, 8, 128, 128])]; + tensor var_4675 = reshape(shape = var_4674, x = value_states_57)[name = string("op_4675")]; + tensor var_4680 = const()[name = string("op_4680"), val = tensor([0, 1, 3, 2])]; + int32 var_4691 = const()[name = string("op_4691"), val = int32(-1)]; + fp16 const_244_promoted = const()[name = string("const_244_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_75 = transpose(perm = var_4658, x = var_4653)[name = string("transpose_61")]; + tensor var_4693 = mul(x = hidden_states_75, y = const_244_promoted)[name = string("op_4693")]; + bool input_131_interleave_0 = const()[name = string("input_131_interleave_0"), val = bool(false)]; + tensor input_131 = concat(axis = var_4691, interleave = input_131_interleave_0, values = (hidden_states_75, var_4693))[name = string("input_131")]; + tensor normed_117_axes_0 = const()[name = string("normed_117_axes_0"), val = tensor([-1])]; + fp16 var_4688_to_fp16 = const()[name = string("op_4688_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_117_cast_fp16 = layer_norm(axes = normed_117_axes_0, epsilon = var_4688_to_fp16, x = input_131)[name = string("normed_117_cast_fp16")]; + tensor normed_119_begin_0 = const()[name = string("normed_119_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_119_end_0 = const()[name = string("normed_119_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_119_end_mask_0 = const()[name = string("normed_119_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_119 = slice_by_index(begin = normed_119_begin_0, end = normed_119_end_0, end_mask = normed_119_end_mask_0, x = normed_117_cast_fp16)[name = string("normed_119")]; + tensor const_247 = const()[name = string("const_247"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693842816)))]; + tensor q_15 = mul(x = normed_119, y = const_247)[name = string("q_15")]; + int32 var_4716 = const()[name = string("op_4716"), val = int32(-1)]; + fp16 const_248_promoted = const()[name = string("const_248_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_77 = transpose(perm = var_4669, x = var_4664)[name = string("transpose_60")]; + tensor var_4718 = mul(x = hidden_states_77, y = const_248_promoted)[name = string("op_4718")]; + bool input_133_interleave_0 = const()[name = string("input_133_interleave_0"), val = bool(false)]; + tensor input_133 = concat(axis = var_4716, interleave = input_133_interleave_0, values = (hidden_states_77, var_4718))[name = string("input_133")]; + tensor normed_121_axes_0 = const()[name = string("normed_121_axes_0"), val = tensor([-1])]; + fp16 var_4713_to_fp16 = const()[name = string("op_4713_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_121_cast_fp16 = layer_norm(axes = normed_121_axes_0, epsilon = var_4713_to_fp16, x = input_133)[name = string("normed_121_cast_fp16")]; + tensor normed_123_begin_0 = const()[name = string("normed_123_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_123_end_0 = const()[name = string("normed_123_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_123_end_mask_0 = const()[name = string("normed_123_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_123 = slice_by_index(begin = normed_123_begin_0, end = normed_123_end_0, end_mask = normed_123_end_mask_0, x = normed_121_cast_fp16)[name = string("normed_123")]; + tensor const_251 = const()[name = string("const_251"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693843136)))]; + tensor k_15 = mul(x = normed_123, y = const_251)[name = string("k_15")]; + tensor var_4744 = mul(x = q_15, y = cos_5)[name = string("op_4744")]; + tensor x1_29_begin_0 = const()[name = string("x1_29_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_29_end_0 = const()[name = string("x1_29_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_29_end_mask_0 = const()[name = string("x1_29_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_29 = slice_by_index(begin = x1_29_begin_0, end = x1_29_end_0, end_mask = x1_29_end_mask_0, x = q_15)[name = string("x1_29")]; + tensor x2_29_begin_0 = const()[name = string("x2_29_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_29_end_0 = const()[name = string("x2_29_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_29_end_mask_0 = const()[name = string("x2_29_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_29 = slice_by_index(begin = x2_29_begin_0, end = x2_29_end_0, end_mask = x2_29_end_mask_0, x = q_15)[name = string("x2_29")]; + fp16 const_254_promoted = const()[name = string("const_254_promoted"), val = fp16(-0x1p+0)]; + tensor var_4765 = mul(x = x2_29, y = const_254_promoted)[name = string("op_4765")]; + int32 var_4767 = const()[name = string("op_4767"), val = int32(-1)]; + bool var_4768_interleave_0 = const()[name = string("op_4768_interleave_0"), val = bool(false)]; + tensor var_4768 = concat(axis = var_4767, interleave = var_4768_interleave_0, values = (var_4765, x1_29))[name = string("op_4768")]; + tensor var_4769 = mul(x = var_4768, y = sin_5)[name = string("op_4769")]; + tensor query_states_59 = add(x = var_4744, y = var_4769)[name = string("query_states_59")]; + tensor var_4772 = mul(x = k_15, y = cos_5)[name = string("op_4772")]; + tensor x1_31_begin_0 = const()[name = string("x1_31_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_31_end_0 = const()[name = string("x1_31_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_31_end_mask_0 = const()[name = string("x1_31_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_31 = slice_by_index(begin = x1_31_begin_0, end = x1_31_end_0, end_mask = x1_31_end_mask_0, x = k_15)[name = string("x1_31")]; + tensor x2_31_begin_0 = const()[name = string("x2_31_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_31_end_0 = const()[name = string("x2_31_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_31_end_mask_0 = const()[name = string("x2_31_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_31 = slice_by_index(begin = x2_31_begin_0, end = x2_31_end_0, end_mask = x2_31_end_mask_0, x = k_15)[name = string("x2_31")]; + fp16 const_257_promoted = const()[name = string("const_257_promoted"), val = fp16(-0x1p+0)]; + tensor var_4793 = mul(x = x2_31, y = const_257_promoted)[name = string("op_4793")]; + int32 var_4795 = const()[name = string("op_4795"), val = int32(-1)]; + bool var_4796_interleave_0 = const()[name = string("op_4796_interleave_0"), val = bool(false)]; + tensor var_4796 = concat(axis = var_4795, interleave = var_4796_interleave_0, values = (var_4793, x1_31))[name = string("op_4796")]; + tensor var_4797 = mul(x = var_4796, y = sin_5)[name = string("op_4797")]; + tensor key_states_73 = add(x = var_4772, y = var_4797)[name = string("key_states_73")]; + tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([7])]; + tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; + tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; + tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([8])]; + int32 concat_128_axis_0 = const()[name = string("concat_128_axis_0"), val = int32(0)]; + bool concat_128_interleave_0 = const()[name = string("concat_128_interleave_0"), val = bool(false)]; + tensor concat_128 = concat(axis = concat_128_axis_0, interleave = concat_128_interleave_0, values = (expand_dims_84, expand_dims_85, current_pos, expand_dims_87))[name = string("concat_128")]; + tensor concat_129_values1_0 = const()[name = string("concat_129_values1_0"), val = tensor([0])]; + tensor concat_129_values3_0 = const()[name = string("concat_129_values3_0"), val = tensor([0])]; + int32 concat_129_axis_0 = const()[name = string("concat_129_axis_0"), val = int32(0)]; + bool concat_129_interleave_0 = const()[name = string("concat_129_interleave_0"), val = bool(false)]; + tensor concat_129 = concat(axis = concat_129_axis_0, interleave = concat_129_interleave_0, values = (expand_dims_88, concat_129_values1_0, var_1039, concat_129_values3_0))[name = string("concat_129")]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_128, begin_mask = model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0, end = concat_129, end_mask = model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_15_stride_0, update = key_states_73, x = coreml_update_state_41)[name = string("model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_98_write_state")]; + tensor coreml_update_state_42 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_98")]; + tensor expand_dims_90 = const()[name = string("expand_dims_90"), val = tensor([35])]; + tensor expand_dims_91 = const()[name = string("expand_dims_91"), val = tensor([0])]; + tensor expand_dims_93 = const()[name = string("expand_dims_93"), val = tensor([0])]; + tensor expand_dims_94 = const()[name = string("expand_dims_94"), val = tensor([36])]; + int32 concat_132_axis_0 = const()[name = string("concat_132_axis_0"), val = int32(0)]; + bool concat_132_interleave_0 = const()[name = string("concat_132_interleave_0"), val = bool(false)]; + tensor concat_132 = concat(axis = concat_132_axis_0, interleave = concat_132_interleave_0, values = (expand_dims_90, expand_dims_91, current_pos, expand_dims_93))[name = string("concat_132")]; + tensor concat_133_values1_0 = const()[name = string("concat_133_values1_0"), val = tensor([0])]; + tensor concat_133_values3_0 = const()[name = string("concat_133_values3_0"), val = tensor([0])]; + int32 concat_133_axis_0 = const()[name = string("concat_133_axis_0"), val = int32(0)]; + bool concat_133_interleave_0 = const()[name = string("concat_133_interleave_0"), val = bool(false)]; + tensor concat_133 = concat(axis = concat_133_axis_0, interleave = concat_133_interleave_0, values = (expand_dims_94, concat_133_values1_0, var_1039, concat_133_values3_0))[name = string("concat_133")]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_59 = transpose(perm = var_4680, x = var_4675)[name = string("transpose_59")]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_132, begin_mask = model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0, end = concat_133, end_mask = model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_16_stride_0, update = value_states_59, x = coreml_update_state_42)[name = string("model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_99_write_state")]; + tensor coreml_update_state_43 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_99")]; + tensor var_4868_begin_0 = const()[name = string("op_4868_begin_0"), val = tensor([7, 0, 0, 0])]; + tensor var_4868_end_0 = const()[name = string("op_4868_end_0"), val = tensor([8, 8, 1024, 128])]; + tensor var_4868_end_mask_0 = const()[name = string("op_4868_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4868_cast_fp16 = slice_by_index(begin = var_4868_begin_0, end = var_4868_end_0, end_mask = var_4868_end_mask_0, x = coreml_update_state_43)[name = string("op_4868_cast_fp16")]; + tensor K_layer_cache_15_axes_0 = const()[name = string("K_layer_cache_15_axes_0"), val = tensor([0])]; + tensor K_layer_cache_15_cast_fp16 = squeeze(axes = K_layer_cache_15_axes_0, x = var_4868_cast_fp16)[name = string("K_layer_cache_15_cast_fp16")]; + tensor var_4875_begin_0 = const()[name = string("op_4875_begin_0"), val = tensor([35, 0, 0, 0])]; + tensor var_4875_end_0 = const()[name = string("op_4875_end_0"), val = tensor([36, 8, 1024, 128])]; + tensor var_4875_end_mask_0 = const()[name = string("op_4875_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4875_cast_fp16 = slice_by_index(begin = var_4875_begin_0, end = var_4875_end_0, end_mask = var_4875_end_mask_0, x = coreml_update_state_43)[name = string("op_4875_cast_fp16")]; + tensor V_layer_cache_15_axes_0 = const()[name = string("V_layer_cache_15_axes_0"), val = tensor([0])]; + tensor V_layer_cache_15_cast_fp16 = squeeze(axes = V_layer_cache_15_axes_0, x = var_4875_cast_fp16)[name = string("V_layer_cache_15_cast_fp16")]; + tensor x_115_axes_0 = const()[name = string("x_115_axes_0"), val = tensor([1])]; + tensor x_115_cast_fp16 = expand_dims(axes = x_115_axes_0, x = K_layer_cache_15_cast_fp16)[name = string("x_115_cast_fp16")]; + tensor var_4904 = const()[name = string("op_4904"), val = tensor([1, 2, 1, 1])]; + tensor x_117_cast_fp16 = tile(reps = var_4904, x = x_115_cast_fp16)[name = string("x_117_cast_fp16")]; + tensor var_4916 = const()[name = string("op_4916"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_77_cast_fp16 = reshape(shape = var_4916, x = x_117_cast_fp16)[name = string("key_states_77_cast_fp16")]; + tensor x_121_axes_0 = const()[name = string("x_121_axes_0"), val = tensor([1])]; + tensor x_121_cast_fp16 = expand_dims(axes = x_121_axes_0, x = V_layer_cache_15_cast_fp16)[name = string("x_121_cast_fp16")]; + tensor var_4924 = const()[name = string("op_4924"), val = tensor([1, 2, 1, 1])]; + tensor x_123_cast_fp16 = tile(reps = var_4924, x = x_121_cast_fp16)[name = string("x_123_cast_fp16")]; + bool var_4951_transpose_x_0 = const()[name = string("op_4951_transpose_x_0"), val = bool(false)]; + bool var_4951_transpose_y_0 = const()[name = string("op_4951_transpose_y_0"), val = bool(true)]; + tensor var_4951 = matmul(transpose_x = var_4951_transpose_x_0, transpose_y = var_4951_transpose_y_0, x = query_states_59, y = key_states_77_cast_fp16)[name = string("op_4951")]; + fp16 var_4952_to_fp16 = const()[name = string("op_4952_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_29_cast_fp16 = mul(x = var_4951, y = var_4952_to_fp16)[name = string("attn_weights_29_cast_fp16")]; + tensor attn_weights_31_cast_fp16 = add(x = attn_weights_29_cast_fp16, y = causal_mask)[name = string("attn_weights_31_cast_fp16")]; + int32 var_4987 = const()[name = string("op_4987"), val = int32(-1)]; + tensor var_4989_cast_fp16 = softmax(axis = var_4987, x = attn_weights_31_cast_fp16)[name = string("op_4989_cast_fp16")]; + tensor concat_138 = const()[name = string("concat_138"), val = tensor([16, 128, 1024])]; + tensor reshape_21_cast_fp16 = reshape(shape = concat_138, x = var_4989_cast_fp16)[name = string("reshape_21_cast_fp16")]; + tensor concat_139 = const()[name = string("concat_139"), val = tensor([16, 1024, 128])]; + tensor reshape_22_cast_fp16 = reshape(shape = concat_139, x = x_123_cast_fp16)[name = string("reshape_22_cast_fp16")]; + bool matmul_7_transpose_x_0 = const()[name = string("matmul_7_transpose_x_0"), val = bool(false)]; + bool matmul_7_transpose_y_0 = const()[name = string("matmul_7_transpose_y_0"), val = bool(false)]; + tensor matmul_7_cast_fp16 = matmul(transpose_x = matmul_7_transpose_x_0, transpose_y = matmul_7_transpose_y_0, x = reshape_21_cast_fp16, y = reshape_22_cast_fp16)[name = string("matmul_7_cast_fp16")]; + tensor concat_143 = const()[name = string("concat_143"), val = tensor([1, 16, 128, 128])]; + tensor reshape_23_cast_fp16 = reshape(shape = concat_143, x = matmul_7_cast_fp16)[name = string("reshape_23_cast_fp16")]; + tensor var_5001_perm_0 = const()[name = string("op_5001_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_5020 = const()[name = string("op_5020"), val = tensor([1, 128, 2048])]; + tensor var_5001_cast_fp16 = transpose(perm = var_5001_perm_0, x = reshape_23_cast_fp16)[name = string("transpose_58")]; + tensor attn_output_75_cast_fp16 = reshape(shape = var_5020, x = var_5001_cast_fp16)[name = string("attn_output_75_cast_fp16")]; + tensor var_5025 = const()[name = string("op_5025"), val = tensor([0, 2, 1])]; + string var_5041_pad_type_0 = const()[name = string("op_5041_pad_type_0"), val = string("valid")]; + int32 var_5041_groups_0 = const()[name = string("op_5041_groups_0"), val = int32(1)]; + tensor var_5041_strides_0 = const()[name = string("op_5041_strides_0"), val = tensor([1])]; + tensor var_5041_pad_0 = const()[name = string("op_5041_pad_0"), val = tensor([0, 0])]; + tensor var_5041_dilations_0 = const()[name = string("op_5041_dilations_0"), val = tensor([1])]; + tensor squeeze_7_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693843456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698037824))))[name = string("squeeze_7_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_5026_cast_fp16 = transpose(perm = var_5025, x = attn_output_75_cast_fp16)[name = string("transpose_57")]; + tensor var_5041_cast_fp16 = conv(dilations = var_5041_dilations_0, groups = var_5041_groups_0, pad = var_5041_pad_0, pad_type = var_5041_pad_type_0, strides = var_5041_strides_0, weight = squeeze_7_cast_fp16_to_fp32_to_fp16_palettized, x = var_5026_cast_fp16)[name = string("op_5041_cast_fp16")]; + tensor var_5045 = const()[name = string("op_5045"), val = tensor([0, 2, 1])]; + tensor attn_output_79_cast_fp16 = transpose(perm = var_5045, x = var_5041_cast_fp16)[name = string("transpose_56")]; + tensor hidden_states_79_cast_fp16 = add(x = hidden_states_71_cast_fp16, y = attn_output_79_cast_fp16)[name = string("hidden_states_79_cast_fp16")]; + int32 var_5058 = const()[name = string("op_5058"), val = int32(-1)]; + fp16 const_269_promoted_to_fp16 = const()[name = string("const_269_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5060_cast_fp16 = mul(x = hidden_states_79_cast_fp16, y = const_269_promoted_to_fp16)[name = string("op_5060_cast_fp16")]; + bool input_137_interleave_0 = const()[name = string("input_137_interleave_0"), val = bool(false)]; + tensor input_137_cast_fp16 = concat(axis = var_5058, interleave = input_137_interleave_0, values = (hidden_states_79_cast_fp16, var_5060_cast_fp16))[name = string("input_137_cast_fp16")]; + tensor normed_125_axes_0 = const()[name = string("normed_125_axes_0"), val = tensor([-1])]; + fp16 var_5055_to_fp16 = const()[name = string("op_5055_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_125_cast_fp16 = layer_norm(axes = normed_125_axes_0, epsilon = var_5055_to_fp16, x = input_137_cast_fp16)[name = string("normed_125_cast_fp16")]; + tensor normed_127_begin_0 = const()[name = string("normed_127_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_127_end_0 = const()[name = string("normed_127_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_127_end_mask_0 = const()[name = string("normed_127_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_127_cast_fp16 = slice_by_index(begin = normed_127_begin_0, end = normed_127_end_0, end_mask = normed_127_end_mask_0, x = normed_125_cast_fp16)[name = string("normed_127_cast_fp16")]; + tensor const_272_promoted_to_fp16 = const()[name = string("const_272_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698168960)))]; + tensor x_125_cast_fp16 = mul(x = normed_127_cast_fp16, y = const_272_promoted_to_fp16)[name = string("x_125_cast_fp16")]; + tensor var_5085 = const()[name = string("op_5085"), val = tensor([0, 2, 1])]; + tensor input_139_axes_0 = const()[name = string("input_139_axes_0"), val = tensor([2])]; + tensor var_5086 = transpose(perm = var_5085, x = x_125_cast_fp16)[name = string("transpose_55")]; + tensor input_139 = expand_dims(axes = input_139_axes_0, x = var_5086)[name = string("input_139")]; + string input_141_pad_type_0 = const()[name = string("input_141_pad_type_0"), val = string("valid")]; + tensor input_141_strides_0 = const()[name = string("input_141_strides_0"), val = tensor([1, 1])]; + tensor input_141_pad_0 = const()[name = string("input_141_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_141_dilations_0 = const()[name = string("input_141_dilations_0"), val = tensor([1, 1])]; + int32 input_141_groups_0 = const()[name = string("input_141_groups_0"), val = int32(1)]; + tensor input_141 = conv(dilations = input_141_dilations_0, groups = input_141_groups_0, pad = input_141_pad_0, pad_type = input_141_pad_type_0, strides = input_141_strides_0, weight = model_model_layers_7_mlp_gate_proj_weight_palettized, x = input_139)[name = string("input_141")]; + string b_15_pad_type_0 = const()[name = string("b_15_pad_type_0"), val = string("valid")]; + tensor b_15_strides_0 = const()[name = string("b_15_strides_0"), val = tensor([1, 1])]; + tensor b_15_pad_0 = const()[name = string("b_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_15_dilations_0 = const()[name = string("b_15_dilations_0"), val = tensor([1, 1])]; + int32 b_15_groups_0 = const()[name = string("b_15_groups_0"), val = int32(1)]; + tensor b_15 = conv(dilations = b_15_dilations_0, groups = b_15_groups_0, pad = b_15_pad_0, pad_type = b_15_pad_type_0, strides = b_15_strides_0, weight = model_model_layers_7_mlp_up_proj_weight_palettized, x = input_139)[name = string("b_15")]; + tensor c_15 = silu(x = input_141)[name = string("c_15")]; + tensor input_143 = mul(x = c_15, y = b_15)[name = string("input_143")]; + string e_15_pad_type_0 = const()[name = string("e_15_pad_type_0"), val = string("valid")]; + tensor e_15_strides_0 = const()[name = string("e_15_strides_0"), val = tensor([1, 1])]; + tensor e_15_pad_0 = const()[name = string("e_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_15_dilations_0 = const()[name = string("e_15_dilations_0"), val = tensor([1, 1])]; + int32 e_15_groups_0 = const()[name = string("e_15_groups_0"), val = int32(1)]; + tensor e_15 = conv(dilations = e_15_dilations_0, groups = e_15_groups_0, pad = e_15_pad_0, pad_type = e_15_pad_type_0, strides = e_15_strides_0, weight = model_model_layers_7_mlp_down_proj_weight_palettized, x = input_143)[name = string("e_15")]; + tensor var_5108_axes_0 = const()[name = string("op_5108_axes_0"), val = tensor([2])]; + tensor var_5108 = squeeze(axes = var_5108_axes_0, x = e_15)[name = string("op_5108")]; + tensor var_5109 = const()[name = string("op_5109"), val = tensor([0, 2, 1])]; + tensor var_5110 = transpose(perm = var_5109, x = var_5108)[name = string("transpose_54")]; + tensor hidden_states_81_cast_fp16 = add(x = hidden_states_79_cast_fp16, y = var_5110)[name = string("hidden_states_81_cast_fp16")]; + int32 var_5122 = const()[name = string("op_5122"), val = int32(-1)]; + fp16 const_273_promoted_to_fp16 = const()[name = string("const_273_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5124_cast_fp16 = mul(x = hidden_states_81_cast_fp16, y = const_273_promoted_to_fp16)[name = string("op_5124_cast_fp16")]; + bool input_145_interleave_0 = const()[name = string("input_145_interleave_0"), val = bool(false)]; + tensor input_145_cast_fp16 = concat(axis = var_5122, interleave = input_145_interleave_0, values = (hidden_states_81_cast_fp16, var_5124_cast_fp16))[name = string("input_145_cast_fp16")]; + tensor normed_129_axes_0 = const()[name = string("normed_129_axes_0"), val = tensor([-1])]; + fp16 var_5119_to_fp16 = const()[name = string("op_5119_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_129_cast_fp16 = layer_norm(axes = normed_129_axes_0, epsilon = var_5119_to_fp16, x = input_145_cast_fp16)[name = string("normed_129_cast_fp16")]; + tensor normed_131_begin_0 = const()[name = string("normed_131_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_131_end_0 = const()[name = string("normed_131_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_131_end_mask_0 = const()[name = string("normed_131_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_131_cast_fp16 = slice_by_index(begin = normed_131_begin_0, end = normed_131_end_0, end_mask = normed_131_end_mask_0, x = normed_129_cast_fp16)[name = string("normed_131_cast_fp16")]; + tensor const_276_promoted_to_fp16 = const()[name = string("const_276_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698173120)))]; + tensor hidden_states_83_cast_fp16 = mul(x = normed_131_cast_fp16, y = const_276_promoted_to_fp16)[name = string("hidden_states_83_cast_fp16")]; + tensor var_5147 = const()[name = string("op_5147"), val = tensor([0, 2, 1])]; + tensor var_5150_axes_0 = const()[name = string("op_5150_axes_0"), val = tensor([2])]; + tensor var_5148_cast_fp16 = transpose(perm = var_5147, x = hidden_states_83_cast_fp16)[name = string("transpose_53")]; + tensor var_5150_cast_fp16 = expand_dims(axes = var_5150_axes_0, x = var_5148_cast_fp16)[name = string("op_5150_cast_fp16")]; + string query_states_65_pad_type_0 = const()[name = string("query_states_65_pad_type_0"), val = string("valid")]; + tensor query_states_65_strides_0 = const()[name = string("query_states_65_strides_0"), val = tensor([1, 1])]; + tensor query_states_65_pad_0 = const()[name = string("query_states_65_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_65_dilations_0 = const()[name = string("query_states_65_dilations_0"), val = tensor([1, 1])]; + int32 query_states_65_groups_0 = const()[name = string("query_states_65_groups_0"), val = int32(1)]; + tensor query_states_65 = conv(dilations = query_states_65_dilations_0, groups = query_states_65_groups_0, pad = query_states_65_pad_0, pad_type = query_states_65_pad_type_0, strides = query_states_65_strides_0, weight = model_model_layers_8_self_attn_q_proj_weight_palettized, x = var_5150_cast_fp16)[name = string("query_states_65")]; + string key_states_81_pad_type_0 = const()[name = string("key_states_81_pad_type_0"), val = string("valid")]; + tensor key_states_81_strides_0 = const()[name = string("key_states_81_strides_0"), val = tensor([1, 1])]; + tensor key_states_81_pad_0 = const()[name = string("key_states_81_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_81_dilations_0 = const()[name = string("key_states_81_dilations_0"), val = tensor([1, 1])]; + int32 key_states_81_groups_0 = const()[name = string("key_states_81_groups_0"), val = int32(1)]; + tensor key_states_81 = conv(dilations = key_states_81_dilations_0, groups = key_states_81_groups_0, pad = key_states_81_pad_0, pad_type = key_states_81_pad_type_0, strides = key_states_81_strides_0, weight = model_model_layers_8_self_attn_k_proj_weight_palettized, x = var_5150_cast_fp16)[name = string("key_states_81")]; + string value_states_65_pad_type_0 = const()[name = string("value_states_65_pad_type_0"), val = string("valid")]; + tensor value_states_65_strides_0 = const()[name = string("value_states_65_strides_0"), val = tensor([1, 1])]; + tensor value_states_65_pad_0 = const()[name = string("value_states_65_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_65_dilations_0 = const()[name = string("value_states_65_dilations_0"), val = tensor([1, 1])]; + int32 value_states_65_groups_0 = const()[name = string("value_states_65_groups_0"), val = int32(1)]; + tensor value_states_65 = conv(dilations = value_states_65_dilations_0, groups = value_states_65_groups_0, pad = value_states_65_pad_0, pad_type = value_states_65_pad_type_0, strides = value_states_65_strides_0, weight = model_model_layers_8_self_attn_v_proj_weight_palettized, x = var_5150_cast_fp16)[name = string("value_states_65")]; + tensor var_5192 = const()[name = string("op_5192"), val = tensor([1, 16, 128, 128])]; + tensor var_5193 = reshape(shape = var_5192, x = query_states_65)[name = string("op_5193")]; + tensor var_5198 = const()[name = string("op_5198"), val = tensor([0, 1, 3, 2])]; + tensor var_5203 = const()[name = string("op_5203"), val = tensor([1, 8, 128, 128])]; + tensor var_5204 = reshape(shape = var_5203, x = key_states_81)[name = string("op_5204")]; + tensor var_5209 = const()[name = string("op_5209"), val = tensor([0, 1, 3, 2])]; + tensor var_5214 = const()[name = string("op_5214"), val = tensor([1, 8, 128, 128])]; + tensor var_5215 = reshape(shape = var_5214, x = value_states_65)[name = string("op_5215")]; + tensor var_5220 = const()[name = string("op_5220"), val = tensor([0, 1, 3, 2])]; + int32 var_5231 = const()[name = string("op_5231"), val = int32(-1)]; + fp16 const_278_promoted = const()[name = string("const_278_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_85 = transpose(perm = var_5198, x = var_5193)[name = string("transpose_52")]; + tensor var_5233 = mul(x = hidden_states_85, y = const_278_promoted)[name = string("op_5233")]; + bool input_149_interleave_0 = const()[name = string("input_149_interleave_0"), val = bool(false)]; + tensor input_149 = concat(axis = var_5231, interleave = input_149_interleave_0, values = (hidden_states_85, var_5233))[name = string("input_149")]; + tensor normed_133_axes_0 = const()[name = string("normed_133_axes_0"), val = tensor([-1])]; + fp16 var_5228_to_fp16 = const()[name = string("op_5228_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_133_cast_fp16 = layer_norm(axes = normed_133_axes_0, epsilon = var_5228_to_fp16, x = input_149)[name = string("normed_133_cast_fp16")]; + tensor normed_135_begin_0 = const()[name = string("normed_135_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_135_end_0 = const()[name = string("normed_135_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_135_end_mask_0 = const()[name = string("normed_135_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_135 = slice_by_index(begin = normed_135_begin_0, end = normed_135_end_0, end_mask = normed_135_end_mask_0, x = normed_133_cast_fp16)[name = string("normed_135")]; + tensor const_281 = const()[name = string("const_281"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698177280)))]; + tensor q_17 = mul(x = normed_135, y = const_281)[name = string("q_17")]; + int32 var_5256 = const()[name = string("op_5256"), val = int32(-1)]; + fp16 const_282_promoted = const()[name = string("const_282_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_87 = transpose(perm = var_5209, x = var_5204)[name = string("transpose_51")]; + tensor var_5258 = mul(x = hidden_states_87, y = const_282_promoted)[name = string("op_5258")]; + bool input_151_interleave_0 = const()[name = string("input_151_interleave_0"), val = bool(false)]; + tensor input_151 = concat(axis = var_5256, interleave = input_151_interleave_0, values = (hidden_states_87, var_5258))[name = string("input_151")]; + tensor normed_137_axes_0 = const()[name = string("normed_137_axes_0"), val = tensor([-1])]; + fp16 var_5253_to_fp16 = const()[name = string("op_5253_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_137_cast_fp16 = layer_norm(axes = normed_137_axes_0, epsilon = var_5253_to_fp16, x = input_151)[name = string("normed_137_cast_fp16")]; + tensor normed_139_begin_0 = const()[name = string("normed_139_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_139_end_0 = const()[name = string("normed_139_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_139_end_mask_0 = const()[name = string("normed_139_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_139 = slice_by_index(begin = normed_139_begin_0, end = normed_139_end_0, end_mask = normed_139_end_mask_0, x = normed_137_cast_fp16)[name = string("normed_139")]; + tensor const_285 = const()[name = string("const_285"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698177600)))]; + tensor k_17 = mul(x = normed_139, y = const_285)[name = string("k_17")]; + tensor var_5284 = mul(x = q_17, y = cos_5)[name = string("op_5284")]; + tensor x1_33_begin_0 = const()[name = string("x1_33_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_33_end_0 = const()[name = string("x1_33_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_33_end_mask_0 = const()[name = string("x1_33_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_33 = slice_by_index(begin = x1_33_begin_0, end = x1_33_end_0, end_mask = x1_33_end_mask_0, x = q_17)[name = string("x1_33")]; + tensor x2_33_begin_0 = const()[name = string("x2_33_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_33_end_0 = const()[name = string("x2_33_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_33_end_mask_0 = const()[name = string("x2_33_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_33 = slice_by_index(begin = x2_33_begin_0, end = x2_33_end_0, end_mask = x2_33_end_mask_0, x = q_17)[name = string("x2_33")]; + fp16 const_288_promoted = const()[name = string("const_288_promoted"), val = fp16(-0x1p+0)]; + tensor var_5305 = mul(x = x2_33, y = const_288_promoted)[name = string("op_5305")]; + int32 var_5307 = const()[name = string("op_5307"), val = int32(-1)]; + bool var_5308_interleave_0 = const()[name = string("op_5308_interleave_0"), val = bool(false)]; + tensor var_5308 = concat(axis = var_5307, interleave = var_5308_interleave_0, values = (var_5305, x1_33))[name = string("op_5308")]; + tensor var_5309 = mul(x = var_5308, y = sin_5)[name = string("op_5309")]; + tensor query_states_67 = add(x = var_5284, y = var_5309)[name = string("query_states_67")]; + tensor var_5312 = mul(x = k_17, y = cos_5)[name = string("op_5312")]; + tensor x1_35_begin_0 = const()[name = string("x1_35_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_35_end_0 = const()[name = string("x1_35_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_35_end_mask_0 = const()[name = string("x1_35_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_35 = slice_by_index(begin = x1_35_begin_0, end = x1_35_end_0, end_mask = x1_35_end_mask_0, x = k_17)[name = string("x1_35")]; + tensor x2_35_begin_0 = const()[name = string("x2_35_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_35_end_0 = const()[name = string("x2_35_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_35_end_mask_0 = const()[name = string("x2_35_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_35 = slice_by_index(begin = x2_35_begin_0, end = x2_35_end_0, end_mask = x2_35_end_mask_0, x = k_17)[name = string("x2_35")]; + fp16 const_291_promoted = const()[name = string("const_291_promoted"), val = fp16(-0x1p+0)]; + tensor var_5333 = mul(x = x2_35, y = const_291_promoted)[name = string("op_5333")]; + int32 var_5335 = const()[name = string("op_5335"), val = int32(-1)]; + bool var_5336_interleave_0 = const()[name = string("op_5336_interleave_0"), val = bool(false)]; + tensor var_5336 = concat(axis = var_5335, interleave = var_5336_interleave_0, values = (var_5333, x1_35))[name = string("op_5336")]; + tensor var_5337 = mul(x = var_5336, y = sin_5)[name = string("op_5337")]; + tensor key_states_83 = add(x = var_5312, y = var_5337)[name = string("key_states_83")]; + tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([8])]; + tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; + tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; + tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([9])]; + int32 concat_146_axis_0 = const()[name = string("concat_146_axis_0"), val = int32(0)]; + bool concat_146_interleave_0 = const()[name = string("concat_146_interleave_0"), val = bool(false)]; + tensor concat_146 = concat(axis = concat_146_axis_0, interleave = concat_146_interleave_0, values = (expand_dims_96, expand_dims_97, current_pos, expand_dims_99))[name = string("concat_146")]; + tensor concat_147_values1_0 = const()[name = string("concat_147_values1_0"), val = tensor([0])]; + tensor concat_147_values3_0 = const()[name = string("concat_147_values3_0"), val = tensor([0])]; + int32 concat_147_axis_0 = const()[name = string("concat_147_axis_0"), val = int32(0)]; + bool concat_147_interleave_0 = const()[name = string("concat_147_interleave_0"), val = bool(false)]; + tensor concat_147 = concat(axis = concat_147_axis_0, interleave = concat_147_interleave_0, values = (expand_dims_100, concat_147_values1_0, var_1039, concat_147_values3_0))[name = string("concat_147")]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_146, begin_mask = model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0, end = concat_147, end_mask = model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_17_stride_0, update = key_states_83, x = coreml_update_state_43)[name = string("model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_100_write_state")]; + tensor coreml_update_state_44 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_100")]; + tensor expand_dims_102 = const()[name = string("expand_dims_102"), val = tensor([36])]; + tensor expand_dims_103 = const()[name = string("expand_dims_103"), val = tensor([0])]; + tensor expand_dims_105 = const()[name = string("expand_dims_105"), val = tensor([0])]; + tensor expand_dims_106 = const()[name = string("expand_dims_106"), val = tensor([37])]; + int32 concat_150_axis_0 = const()[name = string("concat_150_axis_0"), val = int32(0)]; + bool concat_150_interleave_0 = const()[name = string("concat_150_interleave_0"), val = bool(false)]; + tensor concat_150 = concat(axis = concat_150_axis_0, interleave = concat_150_interleave_0, values = (expand_dims_102, expand_dims_103, current_pos, expand_dims_105))[name = string("concat_150")]; + tensor concat_151_values1_0 = const()[name = string("concat_151_values1_0"), val = tensor([0])]; + tensor concat_151_values3_0 = const()[name = string("concat_151_values3_0"), val = tensor([0])]; + int32 concat_151_axis_0 = const()[name = string("concat_151_axis_0"), val = int32(0)]; + bool concat_151_interleave_0 = const()[name = string("concat_151_interleave_0"), val = bool(false)]; + tensor concat_151 = concat(axis = concat_151_axis_0, interleave = concat_151_interleave_0, values = (expand_dims_106, concat_151_values1_0, var_1039, concat_151_values3_0))[name = string("concat_151")]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_67 = transpose(perm = var_5220, x = var_5215)[name = string("transpose_50")]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_150, begin_mask = model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0, end = concat_151, end_mask = model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_18_stride_0, update = value_states_67, x = coreml_update_state_44)[name = string("model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_101_write_state")]; + tensor coreml_update_state_45 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_101")]; + tensor var_5408_begin_0 = const()[name = string("op_5408_begin_0"), val = tensor([8, 0, 0, 0])]; + tensor var_5408_end_0 = const()[name = string("op_5408_end_0"), val = tensor([9, 8, 1024, 128])]; + tensor var_5408_end_mask_0 = const()[name = string("op_5408_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_5408_cast_fp16 = slice_by_index(begin = var_5408_begin_0, end = var_5408_end_0, end_mask = var_5408_end_mask_0, x = coreml_update_state_45)[name = string("op_5408_cast_fp16")]; + tensor K_layer_cache_17_axes_0 = const()[name = string("K_layer_cache_17_axes_0"), val = tensor([0])]; + tensor K_layer_cache_17_cast_fp16 = squeeze(axes = K_layer_cache_17_axes_0, x = var_5408_cast_fp16)[name = string("K_layer_cache_17_cast_fp16")]; + tensor var_5415_begin_0 = const()[name = string("op_5415_begin_0"), val = tensor([36, 0, 0, 0])]; + tensor var_5415_end_0 = const()[name = string("op_5415_end_0"), val = tensor([37, 8, 1024, 128])]; + tensor var_5415_end_mask_0 = const()[name = string("op_5415_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_5415_cast_fp16 = slice_by_index(begin = var_5415_begin_0, end = var_5415_end_0, end_mask = var_5415_end_mask_0, x = coreml_update_state_45)[name = string("op_5415_cast_fp16")]; + tensor V_layer_cache_17_axes_0 = const()[name = string("V_layer_cache_17_axes_0"), val = tensor([0])]; + tensor V_layer_cache_17_cast_fp16 = squeeze(axes = V_layer_cache_17_axes_0, x = var_5415_cast_fp16)[name = string("V_layer_cache_17_cast_fp16")]; + tensor x_131_axes_0 = const()[name = string("x_131_axes_0"), val = tensor([1])]; + tensor x_131_cast_fp16 = expand_dims(axes = x_131_axes_0, x = K_layer_cache_17_cast_fp16)[name = string("x_131_cast_fp16")]; + tensor var_5444 = const()[name = string("op_5444"), val = tensor([1, 2, 1, 1])]; + tensor x_133_cast_fp16 = tile(reps = var_5444, x = x_131_cast_fp16)[name = string("x_133_cast_fp16")]; + tensor var_5456 = const()[name = string("op_5456"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_87_cast_fp16 = reshape(shape = var_5456, x = x_133_cast_fp16)[name = string("key_states_87_cast_fp16")]; + tensor x_137_axes_0 = const()[name = string("x_137_axes_0"), val = tensor([1])]; + tensor x_137_cast_fp16 = expand_dims(axes = x_137_axes_0, x = V_layer_cache_17_cast_fp16)[name = string("x_137_cast_fp16")]; + tensor var_5464 = const()[name = string("op_5464"), val = tensor([1, 2, 1, 1])]; + tensor x_139_cast_fp16 = tile(reps = var_5464, x = x_137_cast_fp16)[name = string("x_139_cast_fp16")]; + bool var_5491_transpose_x_0 = const()[name = string("op_5491_transpose_x_0"), val = bool(false)]; + bool var_5491_transpose_y_0 = const()[name = string("op_5491_transpose_y_0"), val = bool(true)]; + tensor var_5491 = matmul(transpose_x = var_5491_transpose_x_0, transpose_y = var_5491_transpose_y_0, x = query_states_67, y = key_states_87_cast_fp16)[name = string("op_5491")]; + fp16 var_5492_to_fp16 = const()[name = string("op_5492_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_33_cast_fp16 = mul(x = var_5491, y = var_5492_to_fp16)[name = string("attn_weights_33_cast_fp16")]; + tensor attn_weights_35_cast_fp16 = add(x = attn_weights_33_cast_fp16, y = causal_mask)[name = string("attn_weights_35_cast_fp16")]; + int32 var_5527 = const()[name = string("op_5527"), val = int32(-1)]; + tensor var_5529_cast_fp16 = softmax(axis = var_5527, x = attn_weights_35_cast_fp16)[name = string("op_5529_cast_fp16")]; + tensor concat_156 = const()[name = string("concat_156"), val = tensor([16, 128, 1024])]; + tensor reshape_24_cast_fp16 = reshape(shape = concat_156, x = var_5529_cast_fp16)[name = string("reshape_24_cast_fp16")]; + tensor concat_157 = const()[name = string("concat_157"), val = tensor([16, 1024, 128])]; + tensor reshape_25_cast_fp16 = reshape(shape = concat_157, x = x_139_cast_fp16)[name = string("reshape_25_cast_fp16")]; + bool matmul_8_transpose_x_0 = const()[name = string("matmul_8_transpose_x_0"), val = bool(false)]; + bool matmul_8_transpose_y_0 = const()[name = string("matmul_8_transpose_y_0"), val = bool(false)]; + tensor matmul_8_cast_fp16 = matmul(transpose_x = matmul_8_transpose_x_0, transpose_y = matmul_8_transpose_y_0, x = reshape_24_cast_fp16, y = reshape_25_cast_fp16)[name = string("matmul_8_cast_fp16")]; + tensor concat_161 = const()[name = string("concat_161"), val = tensor([1, 16, 128, 128])]; + tensor reshape_26_cast_fp16 = reshape(shape = concat_161, x = matmul_8_cast_fp16)[name = string("reshape_26_cast_fp16")]; + tensor var_5541_perm_0 = const()[name = string("op_5541_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_5560 = const()[name = string("op_5560"), val = tensor([1, 128, 2048])]; + tensor var_5541_cast_fp16 = transpose(perm = var_5541_perm_0, x = reshape_26_cast_fp16)[name = string("transpose_49")]; + tensor attn_output_85_cast_fp16 = reshape(shape = var_5560, x = var_5541_cast_fp16)[name = string("attn_output_85_cast_fp16")]; + tensor var_5565 = const()[name = string("op_5565"), val = tensor([0, 2, 1])]; + string var_5581_pad_type_0 = const()[name = string("op_5581_pad_type_0"), val = string("valid")]; + int32 var_5581_groups_0 = const()[name = string("op_5581_groups_0"), val = int32(1)]; + tensor var_5581_strides_0 = const()[name = string("op_5581_strides_0"), val = tensor([1])]; + tensor var_5581_pad_0 = const()[name = string("op_5581_pad_0"), val = tensor([0, 0])]; + tensor var_5581_dilations_0 = const()[name = string("op_5581_dilations_0"), val = tensor([1])]; + tensor squeeze_8_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698177920))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702372288))))[name = string("squeeze_8_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_5566_cast_fp16 = transpose(perm = var_5565, x = attn_output_85_cast_fp16)[name = string("transpose_48")]; + tensor var_5581_cast_fp16 = conv(dilations = var_5581_dilations_0, groups = var_5581_groups_0, pad = var_5581_pad_0, pad_type = var_5581_pad_type_0, strides = var_5581_strides_0, weight = squeeze_8_cast_fp16_to_fp32_to_fp16_palettized, x = var_5566_cast_fp16)[name = string("op_5581_cast_fp16")]; + tensor var_5585 = const()[name = string("op_5585"), val = tensor([0, 2, 1])]; + tensor attn_output_89_cast_fp16 = transpose(perm = var_5585, x = var_5581_cast_fp16)[name = string("transpose_47")]; + tensor hidden_states_89_cast_fp16 = add(x = hidden_states_81_cast_fp16, y = attn_output_89_cast_fp16)[name = string("hidden_states_89_cast_fp16")]; + int32 var_5598 = const()[name = string("op_5598"), val = int32(-1)]; + fp16 const_303_promoted_to_fp16 = const()[name = string("const_303_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5600_cast_fp16 = mul(x = hidden_states_89_cast_fp16, y = const_303_promoted_to_fp16)[name = string("op_5600_cast_fp16")]; + bool input_155_interleave_0 = const()[name = string("input_155_interleave_0"), val = bool(false)]; + tensor input_155_cast_fp16 = concat(axis = var_5598, interleave = input_155_interleave_0, values = (hidden_states_89_cast_fp16, var_5600_cast_fp16))[name = string("input_155_cast_fp16")]; + tensor normed_141_axes_0 = const()[name = string("normed_141_axes_0"), val = tensor([-1])]; + fp16 var_5595_to_fp16 = const()[name = string("op_5595_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_141_cast_fp16 = layer_norm(axes = normed_141_axes_0, epsilon = var_5595_to_fp16, x = input_155_cast_fp16)[name = string("normed_141_cast_fp16")]; + tensor normed_143_begin_0 = const()[name = string("normed_143_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_143_end_0 = const()[name = string("normed_143_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_143_end_mask_0 = const()[name = string("normed_143_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_143_cast_fp16 = slice_by_index(begin = normed_143_begin_0, end = normed_143_end_0, end_mask = normed_143_end_mask_0, x = normed_141_cast_fp16)[name = string("normed_143_cast_fp16")]; + tensor const_306_promoted_to_fp16 = const()[name = string("const_306_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702503424)))]; + tensor x_141_cast_fp16 = mul(x = normed_143_cast_fp16, y = const_306_promoted_to_fp16)[name = string("x_141_cast_fp16")]; + tensor var_5625 = const()[name = string("op_5625"), val = tensor([0, 2, 1])]; + tensor input_157_axes_0 = const()[name = string("input_157_axes_0"), val = tensor([2])]; + tensor var_5626 = transpose(perm = var_5625, x = x_141_cast_fp16)[name = string("transpose_46")]; + tensor input_157 = expand_dims(axes = input_157_axes_0, x = var_5626)[name = string("input_157")]; + string input_159_pad_type_0 = const()[name = string("input_159_pad_type_0"), val = string("valid")]; + tensor input_159_strides_0 = const()[name = string("input_159_strides_0"), val = tensor([1, 1])]; + tensor input_159_pad_0 = const()[name = string("input_159_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_159_dilations_0 = const()[name = string("input_159_dilations_0"), val = tensor([1, 1])]; + int32 input_159_groups_0 = const()[name = string("input_159_groups_0"), val = int32(1)]; + tensor input_159 = conv(dilations = input_159_dilations_0, groups = input_159_groups_0, pad = input_159_pad_0, pad_type = input_159_pad_type_0, strides = input_159_strides_0, weight = model_model_layers_8_mlp_gate_proj_weight_palettized, x = input_157)[name = string("input_159")]; + string b_17_pad_type_0 = const()[name = string("b_17_pad_type_0"), val = string("valid")]; + tensor b_17_strides_0 = const()[name = string("b_17_strides_0"), val = tensor([1, 1])]; + tensor b_17_pad_0 = const()[name = string("b_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_17_dilations_0 = const()[name = string("b_17_dilations_0"), val = tensor([1, 1])]; + int32 b_17_groups_0 = const()[name = string("b_17_groups_0"), val = int32(1)]; + tensor b_17 = conv(dilations = b_17_dilations_0, groups = b_17_groups_0, pad = b_17_pad_0, pad_type = b_17_pad_type_0, strides = b_17_strides_0, weight = model_model_layers_8_mlp_up_proj_weight_palettized, x = input_157)[name = string("b_17")]; + tensor c_17 = silu(x = input_159)[name = string("c_17")]; + tensor input_161 = mul(x = c_17, y = b_17)[name = string("input_161")]; + string e_17_pad_type_0 = const()[name = string("e_17_pad_type_0"), val = string("valid")]; + tensor e_17_strides_0 = const()[name = string("e_17_strides_0"), val = tensor([1, 1])]; + tensor e_17_pad_0 = const()[name = string("e_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_17_dilations_0 = const()[name = string("e_17_dilations_0"), val = tensor([1, 1])]; + int32 e_17_groups_0 = const()[name = string("e_17_groups_0"), val = int32(1)]; + tensor e_17 = conv(dilations = e_17_dilations_0, groups = e_17_groups_0, pad = e_17_pad_0, pad_type = e_17_pad_type_0, strides = e_17_strides_0, weight = model_model_layers_8_mlp_down_proj_weight_palettized, x = input_161)[name = string("e_17")]; + tensor var_5648_axes_0 = const()[name = string("op_5648_axes_0"), val = tensor([2])]; + tensor var_5648 = squeeze(axes = var_5648_axes_0, x = e_17)[name = string("op_5648")]; + tensor var_5649 = const()[name = string("op_5649"), val = tensor([0, 2, 1])]; + tensor var_5650 = transpose(perm = var_5649, x = var_5648)[name = string("transpose_45")]; + tensor hidden_states_91_cast_fp16 = add(x = hidden_states_89_cast_fp16, y = var_5650)[name = string("hidden_states_91_cast_fp16")]; + int32 var_5662 = const()[name = string("op_5662"), val = int32(-1)]; + fp16 const_307_promoted_to_fp16 = const()[name = string("const_307_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5664_cast_fp16 = mul(x = hidden_states_91_cast_fp16, y = const_307_promoted_to_fp16)[name = string("op_5664_cast_fp16")]; + bool input_163_interleave_0 = const()[name = string("input_163_interleave_0"), val = bool(false)]; + tensor input_163_cast_fp16 = concat(axis = var_5662, interleave = input_163_interleave_0, values = (hidden_states_91_cast_fp16, var_5664_cast_fp16))[name = string("input_163_cast_fp16")]; + tensor normed_145_axes_0 = const()[name = string("normed_145_axes_0"), val = tensor([-1])]; + fp16 var_5659_to_fp16 = const()[name = string("op_5659_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_145_cast_fp16 = layer_norm(axes = normed_145_axes_0, epsilon = var_5659_to_fp16, x = input_163_cast_fp16)[name = string("normed_145_cast_fp16")]; + tensor normed_147_begin_0 = const()[name = string("normed_147_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_147_end_0 = const()[name = string("normed_147_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_147_end_mask_0 = const()[name = string("normed_147_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_147_cast_fp16 = slice_by_index(begin = normed_147_begin_0, end = normed_147_end_0, end_mask = normed_147_end_mask_0, x = normed_145_cast_fp16)[name = string("normed_147_cast_fp16")]; + tensor const_310_promoted_to_fp16 = const()[name = string("const_310_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702507584)))]; + tensor hidden_states_93_cast_fp16 = mul(x = normed_147_cast_fp16, y = const_310_promoted_to_fp16)[name = string("hidden_states_93_cast_fp16")]; + tensor var_5687 = const()[name = string("op_5687"), val = tensor([0, 2, 1])]; + tensor var_5690_axes_0 = const()[name = string("op_5690_axes_0"), val = tensor([2])]; + tensor var_5688_cast_fp16 = transpose(perm = var_5687, x = hidden_states_93_cast_fp16)[name = string("transpose_44")]; + tensor var_5690_cast_fp16 = expand_dims(axes = var_5690_axes_0, x = var_5688_cast_fp16)[name = string("op_5690_cast_fp16")]; + string query_states_73_pad_type_0 = const()[name = string("query_states_73_pad_type_0"), val = string("valid")]; + tensor query_states_73_strides_0 = const()[name = string("query_states_73_strides_0"), val = tensor([1, 1])]; + tensor query_states_73_pad_0 = const()[name = string("query_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_73_dilations_0 = const()[name = string("query_states_73_dilations_0"), val = tensor([1, 1])]; + int32 query_states_73_groups_0 = const()[name = string("query_states_73_groups_0"), val = int32(1)]; + tensor query_states_73 = conv(dilations = query_states_73_dilations_0, groups = query_states_73_groups_0, pad = query_states_73_pad_0, pad_type = query_states_73_pad_type_0, strides = query_states_73_strides_0, weight = model_model_layers_9_self_attn_q_proj_weight_palettized, x = var_5690_cast_fp16)[name = string("query_states_73")]; + string key_states_91_pad_type_0 = const()[name = string("key_states_91_pad_type_0"), val = string("valid")]; + tensor key_states_91_strides_0 = const()[name = string("key_states_91_strides_0"), val = tensor([1, 1])]; + tensor key_states_91_pad_0 = const()[name = string("key_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_91_dilations_0 = const()[name = string("key_states_91_dilations_0"), val = tensor([1, 1])]; + int32 key_states_91_groups_0 = const()[name = string("key_states_91_groups_0"), val = int32(1)]; + tensor key_states_91 = conv(dilations = key_states_91_dilations_0, groups = key_states_91_groups_0, pad = key_states_91_pad_0, pad_type = key_states_91_pad_type_0, strides = key_states_91_strides_0, weight = model_model_layers_9_self_attn_k_proj_weight_palettized, x = var_5690_cast_fp16)[name = string("key_states_91")]; + string value_states_73_pad_type_0 = const()[name = string("value_states_73_pad_type_0"), val = string("valid")]; + tensor value_states_73_strides_0 = const()[name = string("value_states_73_strides_0"), val = tensor([1, 1])]; + tensor value_states_73_pad_0 = const()[name = string("value_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_73_dilations_0 = const()[name = string("value_states_73_dilations_0"), val = tensor([1, 1])]; + int32 value_states_73_groups_0 = const()[name = string("value_states_73_groups_0"), val = int32(1)]; + tensor value_states_73 = conv(dilations = value_states_73_dilations_0, groups = value_states_73_groups_0, pad = value_states_73_pad_0, pad_type = value_states_73_pad_type_0, strides = value_states_73_strides_0, weight = model_model_layers_9_self_attn_v_proj_weight_palettized, x = var_5690_cast_fp16)[name = string("value_states_73")]; + tensor var_5732 = const()[name = string("op_5732"), val = tensor([1, 16, 128, 128])]; + tensor var_5733 = reshape(shape = var_5732, x = query_states_73)[name = string("op_5733")]; + tensor var_5738 = const()[name = string("op_5738"), val = tensor([0, 1, 3, 2])]; + tensor var_5743 = const()[name = string("op_5743"), val = tensor([1, 8, 128, 128])]; + tensor var_5744 = reshape(shape = var_5743, x = key_states_91)[name = string("op_5744")]; + tensor var_5749 = const()[name = string("op_5749"), val = tensor([0, 1, 3, 2])]; + tensor var_5754 = const()[name = string("op_5754"), val = tensor([1, 8, 128, 128])]; + tensor var_5755 = reshape(shape = var_5754, x = value_states_73)[name = string("op_5755")]; + tensor var_5760 = const()[name = string("op_5760"), val = tensor([0, 1, 3, 2])]; + int32 var_5771 = const()[name = string("op_5771"), val = int32(-1)]; + fp16 const_312_promoted = const()[name = string("const_312_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_95 = transpose(perm = var_5738, x = var_5733)[name = string("transpose_43")]; + tensor var_5773 = mul(x = hidden_states_95, y = const_312_promoted)[name = string("op_5773")]; + bool input_167_interleave_0 = const()[name = string("input_167_interleave_0"), val = bool(false)]; + tensor input_167 = concat(axis = var_5771, interleave = input_167_interleave_0, values = (hidden_states_95, var_5773))[name = string("input_167")]; + tensor normed_149_axes_0 = const()[name = string("normed_149_axes_0"), val = tensor([-1])]; + fp16 var_5768_to_fp16 = const()[name = string("op_5768_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_149_cast_fp16 = layer_norm(axes = normed_149_axes_0, epsilon = var_5768_to_fp16, x = input_167)[name = string("normed_149_cast_fp16")]; + tensor normed_151_begin_0 = const()[name = string("normed_151_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_151_end_0 = const()[name = string("normed_151_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_151_end_mask_0 = const()[name = string("normed_151_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_151 = slice_by_index(begin = normed_151_begin_0, end = normed_151_end_0, end_mask = normed_151_end_mask_0, x = normed_149_cast_fp16)[name = string("normed_151")]; + tensor const_315 = const()[name = string("const_315"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702511744)))]; + tensor q_19 = mul(x = normed_151, y = const_315)[name = string("q_19")]; + int32 var_5796 = const()[name = string("op_5796"), val = int32(-1)]; + fp16 const_316_promoted = const()[name = string("const_316_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_97 = transpose(perm = var_5749, x = var_5744)[name = string("transpose_42")]; + tensor var_5798 = mul(x = hidden_states_97, y = const_316_promoted)[name = string("op_5798")]; + bool input_169_interleave_0 = const()[name = string("input_169_interleave_0"), val = bool(false)]; + tensor input_169 = concat(axis = var_5796, interleave = input_169_interleave_0, values = (hidden_states_97, var_5798))[name = string("input_169")]; + tensor normed_153_axes_0 = const()[name = string("normed_153_axes_0"), val = tensor([-1])]; + fp16 var_5793_to_fp16 = const()[name = string("op_5793_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_153_cast_fp16 = layer_norm(axes = normed_153_axes_0, epsilon = var_5793_to_fp16, x = input_169)[name = string("normed_153_cast_fp16")]; + tensor normed_155_begin_0 = const()[name = string("normed_155_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_155_end_0 = const()[name = string("normed_155_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_155_end_mask_0 = const()[name = string("normed_155_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_155 = slice_by_index(begin = normed_155_begin_0, end = normed_155_end_0, end_mask = normed_155_end_mask_0, x = normed_153_cast_fp16)[name = string("normed_155")]; + tensor const_319 = const()[name = string("const_319"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702512064)))]; + tensor k_19 = mul(x = normed_155, y = const_319)[name = string("k_19")]; + tensor var_5824 = mul(x = q_19, y = cos_5)[name = string("op_5824")]; + tensor x1_37_begin_0 = const()[name = string("x1_37_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_37_end_0 = const()[name = string("x1_37_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_37_end_mask_0 = const()[name = string("x1_37_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_37 = slice_by_index(begin = x1_37_begin_0, end = x1_37_end_0, end_mask = x1_37_end_mask_0, x = q_19)[name = string("x1_37")]; + tensor x2_37_begin_0 = const()[name = string("x2_37_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_37_end_0 = const()[name = string("x2_37_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_37_end_mask_0 = const()[name = string("x2_37_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_37 = slice_by_index(begin = x2_37_begin_0, end = x2_37_end_0, end_mask = x2_37_end_mask_0, x = q_19)[name = string("x2_37")]; + fp16 const_322_promoted = const()[name = string("const_322_promoted"), val = fp16(-0x1p+0)]; + tensor var_5845 = mul(x = x2_37, y = const_322_promoted)[name = string("op_5845")]; + int32 var_5847 = const()[name = string("op_5847"), val = int32(-1)]; + bool var_5848_interleave_0 = const()[name = string("op_5848_interleave_0"), val = bool(false)]; + tensor var_5848 = concat(axis = var_5847, interleave = var_5848_interleave_0, values = (var_5845, x1_37))[name = string("op_5848")]; + tensor var_5849 = mul(x = var_5848, y = sin_5)[name = string("op_5849")]; + tensor query_states_75 = add(x = var_5824, y = var_5849)[name = string("query_states_75")]; + tensor var_5852 = mul(x = k_19, y = cos_5)[name = string("op_5852")]; + tensor x1_39_begin_0 = const()[name = string("x1_39_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_39_end_0 = const()[name = string("x1_39_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_39_end_mask_0 = const()[name = string("x1_39_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_39 = slice_by_index(begin = x1_39_begin_0, end = x1_39_end_0, end_mask = x1_39_end_mask_0, x = k_19)[name = string("x1_39")]; + tensor x2_39_begin_0 = const()[name = string("x2_39_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_39_end_0 = const()[name = string("x2_39_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_39_end_mask_0 = const()[name = string("x2_39_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_39 = slice_by_index(begin = x2_39_begin_0, end = x2_39_end_0, end_mask = x2_39_end_mask_0, x = k_19)[name = string("x2_39")]; + fp16 const_325_promoted = const()[name = string("const_325_promoted"), val = fp16(-0x1p+0)]; + tensor var_5873 = mul(x = x2_39, y = const_325_promoted)[name = string("op_5873")]; + int32 var_5875 = const()[name = string("op_5875"), val = int32(-1)]; + bool var_5876_interleave_0 = const()[name = string("op_5876_interleave_0"), val = bool(false)]; + tensor var_5876 = concat(axis = var_5875, interleave = var_5876_interleave_0, values = (var_5873, x1_39))[name = string("op_5876")]; + tensor var_5877 = mul(x = var_5876, y = sin_5)[name = string("op_5877")]; + tensor key_states_93 = add(x = var_5852, y = var_5877)[name = string("key_states_93")]; + tensor expand_dims_108 = const()[name = string("expand_dims_108"), val = tensor([9])]; + tensor expand_dims_109 = const()[name = string("expand_dims_109"), val = tensor([0])]; + tensor expand_dims_111 = const()[name = string("expand_dims_111"), val = tensor([0])]; + tensor expand_dims_112 = const()[name = string("expand_dims_112"), val = tensor([10])]; + int32 concat_164_axis_0 = const()[name = string("concat_164_axis_0"), val = int32(0)]; + bool concat_164_interleave_0 = const()[name = string("concat_164_interleave_0"), val = bool(false)]; + tensor concat_164 = concat(axis = concat_164_axis_0, interleave = concat_164_interleave_0, values = (expand_dims_108, expand_dims_109, current_pos, expand_dims_111))[name = string("concat_164")]; + tensor concat_165_values1_0 = const()[name = string("concat_165_values1_0"), val = tensor([0])]; + tensor concat_165_values3_0 = const()[name = string("concat_165_values3_0"), val = tensor([0])]; + int32 concat_165_axis_0 = const()[name = string("concat_165_axis_0"), val = int32(0)]; + bool concat_165_interleave_0 = const()[name = string("concat_165_interleave_0"), val = bool(false)]; + tensor concat_165 = concat(axis = concat_165_axis_0, interleave = concat_165_interleave_0, values = (expand_dims_112, concat_165_values1_0, var_1039, concat_165_values3_0))[name = string("concat_165")]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_164, begin_mask = model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0, end = concat_165, end_mask = model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_19_stride_0, update = key_states_93, x = coreml_update_state_45)[name = string("model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_102_write_state")]; + tensor coreml_update_state_46 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_102")]; + tensor expand_dims_114 = const()[name = string("expand_dims_114"), val = tensor([37])]; + tensor expand_dims_115 = const()[name = string("expand_dims_115"), val = tensor([0])]; + tensor expand_dims_117 = const()[name = string("expand_dims_117"), val = tensor([0])]; + tensor expand_dims_118 = const()[name = string("expand_dims_118"), val = tensor([38])]; + int32 concat_168_axis_0 = const()[name = string("concat_168_axis_0"), val = int32(0)]; + bool concat_168_interleave_0 = const()[name = string("concat_168_interleave_0"), val = bool(false)]; + tensor concat_168 = concat(axis = concat_168_axis_0, interleave = concat_168_interleave_0, values = (expand_dims_114, expand_dims_115, current_pos, expand_dims_117))[name = string("concat_168")]; + tensor concat_169_values1_0 = const()[name = string("concat_169_values1_0"), val = tensor([0])]; + tensor concat_169_values3_0 = const()[name = string("concat_169_values3_0"), val = tensor([0])]; + int32 concat_169_axis_0 = const()[name = string("concat_169_axis_0"), val = int32(0)]; + bool concat_169_interleave_0 = const()[name = string("concat_169_interleave_0"), val = bool(false)]; + tensor concat_169 = concat(axis = concat_169_axis_0, interleave = concat_169_interleave_0, values = (expand_dims_118, concat_169_values1_0, var_1039, concat_169_values3_0))[name = string("concat_169")]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_75 = transpose(perm = var_5760, x = var_5755)[name = string("transpose_41")]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_168, begin_mask = model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0, end = concat_169, end_mask = model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_20_stride_0, update = value_states_75, x = coreml_update_state_46)[name = string("model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_103_write_state")]; + tensor coreml_update_state_47 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_103")]; + tensor var_5948_begin_0 = const()[name = string("op_5948_begin_0"), val = tensor([9, 0, 0, 0])]; + tensor var_5948_end_0 = const()[name = string("op_5948_end_0"), val = tensor([10, 8, 1024, 128])]; + tensor var_5948_end_mask_0 = const()[name = string("op_5948_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_5948_cast_fp16 = slice_by_index(begin = var_5948_begin_0, end = var_5948_end_0, end_mask = var_5948_end_mask_0, x = coreml_update_state_47)[name = string("op_5948_cast_fp16")]; + tensor K_layer_cache_19_axes_0 = const()[name = string("K_layer_cache_19_axes_0"), val = tensor([0])]; + tensor K_layer_cache_19_cast_fp16 = squeeze(axes = K_layer_cache_19_axes_0, x = var_5948_cast_fp16)[name = string("K_layer_cache_19_cast_fp16")]; + tensor var_5955_begin_0 = const()[name = string("op_5955_begin_0"), val = tensor([37, 0, 0, 0])]; + tensor var_5955_end_0 = const()[name = string("op_5955_end_0"), val = tensor([38, 8, 1024, 128])]; + tensor var_5955_end_mask_0 = const()[name = string("op_5955_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_5955_cast_fp16 = slice_by_index(begin = var_5955_begin_0, end = var_5955_end_0, end_mask = var_5955_end_mask_0, x = coreml_update_state_47)[name = string("op_5955_cast_fp16")]; + tensor V_layer_cache_19_axes_0 = const()[name = string("V_layer_cache_19_axes_0"), val = tensor([0])]; + tensor V_layer_cache_19_cast_fp16 = squeeze(axes = V_layer_cache_19_axes_0, x = var_5955_cast_fp16)[name = string("V_layer_cache_19_cast_fp16")]; + tensor x_147_axes_0 = const()[name = string("x_147_axes_0"), val = tensor([1])]; + tensor x_147_cast_fp16 = expand_dims(axes = x_147_axes_0, x = K_layer_cache_19_cast_fp16)[name = string("x_147_cast_fp16")]; + tensor var_5984 = const()[name = string("op_5984"), val = tensor([1, 2, 1, 1])]; + tensor x_149_cast_fp16 = tile(reps = var_5984, x = x_147_cast_fp16)[name = string("x_149_cast_fp16")]; + tensor var_5996 = const()[name = string("op_5996"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_97_cast_fp16 = reshape(shape = var_5996, x = x_149_cast_fp16)[name = string("key_states_97_cast_fp16")]; + tensor x_153_axes_0 = const()[name = string("x_153_axes_0"), val = tensor([1])]; + tensor x_153_cast_fp16 = expand_dims(axes = x_153_axes_0, x = V_layer_cache_19_cast_fp16)[name = string("x_153_cast_fp16")]; + tensor var_6004 = const()[name = string("op_6004"), val = tensor([1, 2, 1, 1])]; + tensor x_155_cast_fp16 = tile(reps = var_6004, x = x_153_cast_fp16)[name = string("x_155_cast_fp16")]; + bool var_6031_transpose_x_0 = const()[name = string("op_6031_transpose_x_0"), val = bool(false)]; + bool var_6031_transpose_y_0 = const()[name = string("op_6031_transpose_y_0"), val = bool(true)]; + tensor var_6031 = matmul(transpose_x = var_6031_transpose_x_0, transpose_y = var_6031_transpose_y_0, x = query_states_75, y = key_states_97_cast_fp16)[name = string("op_6031")]; + fp16 var_6032_to_fp16 = const()[name = string("op_6032_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_37_cast_fp16 = mul(x = var_6031, y = var_6032_to_fp16)[name = string("attn_weights_37_cast_fp16")]; + tensor attn_weights_39_cast_fp16 = add(x = attn_weights_37_cast_fp16, y = causal_mask)[name = string("attn_weights_39_cast_fp16")]; + int32 var_6067 = const()[name = string("op_6067"), val = int32(-1)]; + tensor var_6069_cast_fp16 = softmax(axis = var_6067, x = attn_weights_39_cast_fp16)[name = string("op_6069_cast_fp16")]; + tensor concat_174 = const()[name = string("concat_174"), val = tensor([16, 128, 1024])]; + tensor reshape_27_cast_fp16 = reshape(shape = concat_174, x = var_6069_cast_fp16)[name = string("reshape_27_cast_fp16")]; + tensor concat_175 = const()[name = string("concat_175"), val = tensor([16, 1024, 128])]; + tensor reshape_28_cast_fp16 = reshape(shape = concat_175, x = x_155_cast_fp16)[name = string("reshape_28_cast_fp16")]; + bool matmul_9_transpose_x_0 = const()[name = string("matmul_9_transpose_x_0"), val = bool(false)]; + bool matmul_9_transpose_y_0 = const()[name = string("matmul_9_transpose_y_0"), val = bool(false)]; + tensor matmul_9_cast_fp16 = matmul(transpose_x = matmul_9_transpose_x_0, transpose_y = matmul_9_transpose_y_0, x = reshape_27_cast_fp16, y = reshape_28_cast_fp16)[name = string("matmul_9_cast_fp16")]; + tensor concat_179 = const()[name = string("concat_179"), val = tensor([1, 16, 128, 128])]; + tensor reshape_29_cast_fp16 = reshape(shape = concat_179, x = matmul_9_cast_fp16)[name = string("reshape_29_cast_fp16")]; + tensor var_6081_perm_0 = const()[name = string("op_6081_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_6100 = const()[name = string("op_6100"), val = tensor([1, 128, 2048])]; + tensor var_6081_cast_fp16 = transpose(perm = var_6081_perm_0, x = reshape_29_cast_fp16)[name = string("transpose_40")]; + tensor attn_output_95_cast_fp16 = reshape(shape = var_6100, x = var_6081_cast_fp16)[name = string("attn_output_95_cast_fp16")]; + tensor var_6105 = const()[name = string("op_6105"), val = tensor([0, 2, 1])]; + string var_6121_pad_type_0 = const()[name = string("op_6121_pad_type_0"), val = string("valid")]; + int32 var_6121_groups_0 = const()[name = string("op_6121_groups_0"), val = int32(1)]; + tensor var_6121_strides_0 = const()[name = string("op_6121_strides_0"), val = tensor([1])]; + tensor var_6121_pad_0 = const()[name = string("op_6121_pad_0"), val = tensor([0, 0])]; + tensor var_6121_dilations_0 = const()[name = string("op_6121_dilations_0"), val = tensor([1])]; + tensor squeeze_9_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702512384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706706752))))[name = string("squeeze_9_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_6106_cast_fp16 = transpose(perm = var_6105, x = attn_output_95_cast_fp16)[name = string("transpose_39")]; + tensor var_6121_cast_fp16 = conv(dilations = var_6121_dilations_0, groups = var_6121_groups_0, pad = var_6121_pad_0, pad_type = var_6121_pad_type_0, strides = var_6121_strides_0, weight = squeeze_9_cast_fp16_to_fp32_to_fp16_palettized, x = var_6106_cast_fp16)[name = string("op_6121_cast_fp16")]; + tensor var_6125 = const()[name = string("op_6125"), val = tensor([0, 2, 1])]; + tensor attn_output_99_cast_fp16 = transpose(perm = var_6125, x = var_6121_cast_fp16)[name = string("transpose_38")]; + tensor hidden_states_99_cast_fp16 = add(x = hidden_states_91_cast_fp16, y = attn_output_99_cast_fp16)[name = string("hidden_states_99_cast_fp16")]; + int32 var_6138 = const()[name = string("op_6138"), val = int32(-1)]; + fp16 const_337_promoted_to_fp16 = const()[name = string("const_337_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6140_cast_fp16 = mul(x = hidden_states_99_cast_fp16, y = const_337_promoted_to_fp16)[name = string("op_6140_cast_fp16")]; + bool input_173_interleave_0 = const()[name = string("input_173_interleave_0"), val = bool(false)]; + tensor input_173_cast_fp16 = concat(axis = var_6138, interleave = input_173_interleave_0, values = (hidden_states_99_cast_fp16, var_6140_cast_fp16))[name = string("input_173_cast_fp16")]; + tensor normed_157_axes_0 = const()[name = string("normed_157_axes_0"), val = tensor([-1])]; + fp16 var_6135_to_fp16 = const()[name = string("op_6135_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_157_cast_fp16 = layer_norm(axes = normed_157_axes_0, epsilon = var_6135_to_fp16, x = input_173_cast_fp16)[name = string("normed_157_cast_fp16")]; + tensor normed_159_begin_0 = const()[name = string("normed_159_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_159_end_0 = const()[name = string("normed_159_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_159_end_mask_0 = const()[name = string("normed_159_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_159_cast_fp16 = slice_by_index(begin = normed_159_begin_0, end = normed_159_end_0, end_mask = normed_159_end_mask_0, x = normed_157_cast_fp16)[name = string("normed_159_cast_fp16")]; + tensor const_340_promoted_to_fp16 = const()[name = string("const_340_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706837888)))]; + tensor x_157_cast_fp16 = mul(x = normed_159_cast_fp16, y = const_340_promoted_to_fp16)[name = string("x_157_cast_fp16")]; + tensor var_6165 = const()[name = string("op_6165"), val = tensor([0, 2, 1])]; + tensor input_175_axes_0 = const()[name = string("input_175_axes_0"), val = tensor([2])]; + tensor var_6166 = transpose(perm = var_6165, x = x_157_cast_fp16)[name = string("transpose_37")]; + tensor input_175 = expand_dims(axes = input_175_axes_0, x = var_6166)[name = string("input_175")]; + string input_177_pad_type_0 = const()[name = string("input_177_pad_type_0"), val = string("valid")]; + tensor input_177_strides_0 = const()[name = string("input_177_strides_0"), val = tensor([1, 1])]; + tensor input_177_pad_0 = const()[name = string("input_177_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_177_dilations_0 = const()[name = string("input_177_dilations_0"), val = tensor([1, 1])]; + int32 input_177_groups_0 = const()[name = string("input_177_groups_0"), val = int32(1)]; + tensor input_177 = conv(dilations = input_177_dilations_0, groups = input_177_groups_0, pad = input_177_pad_0, pad_type = input_177_pad_type_0, strides = input_177_strides_0, weight = model_model_layers_9_mlp_gate_proj_weight_palettized, x = input_175)[name = string("input_177")]; + string b_19_pad_type_0 = const()[name = string("b_19_pad_type_0"), val = string("valid")]; + tensor b_19_strides_0 = const()[name = string("b_19_strides_0"), val = tensor([1, 1])]; + tensor b_19_pad_0 = const()[name = string("b_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_19_dilations_0 = const()[name = string("b_19_dilations_0"), val = tensor([1, 1])]; + int32 b_19_groups_0 = const()[name = string("b_19_groups_0"), val = int32(1)]; + tensor b_19 = conv(dilations = b_19_dilations_0, groups = b_19_groups_0, pad = b_19_pad_0, pad_type = b_19_pad_type_0, strides = b_19_strides_0, weight = model_model_layers_9_mlp_up_proj_weight_palettized, x = input_175)[name = string("b_19")]; + tensor c_19 = silu(x = input_177)[name = string("c_19")]; + tensor input_179 = mul(x = c_19, y = b_19)[name = string("input_179")]; + string e_19_pad_type_0 = const()[name = string("e_19_pad_type_0"), val = string("valid")]; + tensor e_19_strides_0 = const()[name = string("e_19_strides_0"), val = tensor([1, 1])]; + tensor e_19_pad_0 = const()[name = string("e_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_19_dilations_0 = const()[name = string("e_19_dilations_0"), val = tensor([1, 1])]; + int32 e_19_groups_0 = const()[name = string("e_19_groups_0"), val = int32(1)]; + tensor e_19 = conv(dilations = e_19_dilations_0, groups = e_19_groups_0, pad = e_19_pad_0, pad_type = e_19_pad_type_0, strides = e_19_strides_0, weight = model_model_layers_9_mlp_down_proj_weight_palettized, x = input_179)[name = string("e_19")]; + tensor var_6188_axes_0 = const()[name = string("op_6188_axes_0"), val = tensor([2])]; + tensor var_6188 = squeeze(axes = var_6188_axes_0, x = e_19)[name = string("op_6188")]; + tensor var_6189 = const()[name = string("op_6189"), val = tensor([0, 2, 1])]; + tensor var_6190 = transpose(perm = var_6189, x = var_6188)[name = string("transpose_36")]; + tensor hidden_states_101_cast_fp16 = add(x = hidden_states_99_cast_fp16, y = var_6190)[name = string("hidden_states_101_cast_fp16")]; + int32 var_6202 = const()[name = string("op_6202"), val = int32(-1)]; + fp16 const_341_promoted_to_fp16 = const()[name = string("const_341_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6204_cast_fp16 = mul(x = hidden_states_101_cast_fp16, y = const_341_promoted_to_fp16)[name = string("op_6204_cast_fp16")]; + bool input_181_interleave_0 = const()[name = string("input_181_interleave_0"), val = bool(false)]; + tensor input_181_cast_fp16 = concat(axis = var_6202, interleave = input_181_interleave_0, values = (hidden_states_101_cast_fp16, var_6204_cast_fp16))[name = string("input_181_cast_fp16")]; + tensor normed_161_axes_0 = const()[name = string("normed_161_axes_0"), val = tensor([-1])]; + fp16 var_6199_to_fp16 = const()[name = string("op_6199_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_161_cast_fp16 = layer_norm(axes = normed_161_axes_0, epsilon = var_6199_to_fp16, x = input_181_cast_fp16)[name = string("normed_161_cast_fp16")]; + tensor normed_163_begin_0 = const()[name = string("normed_163_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_163_end_0 = const()[name = string("normed_163_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_163_end_mask_0 = const()[name = string("normed_163_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_163_cast_fp16 = slice_by_index(begin = normed_163_begin_0, end = normed_163_end_0, end_mask = normed_163_end_mask_0, x = normed_161_cast_fp16)[name = string("normed_163_cast_fp16")]; + tensor const_344_promoted_to_fp16 = const()[name = string("const_344_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706842048)))]; + tensor hidden_states_103_cast_fp16 = mul(x = normed_163_cast_fp16, y = const_344_promoted_to_fp16)[name = string("hidden_states_103_cast_fp16")]; + tensor var_6227 = const()[name = string("op_6227"), val = tensor([0, 2, 1])]; + tensor var_6230_axes_0 = const()[name = string("op_6230_axes_0"), val = tensor([2])]; + tensor var_6228_cast_fp16 = transpose(perm = var_6227, x = hidden_states_103_cast_fp16)[name = string("transpose_35")]; + tensor var_6230_cast_fp16 = expand_dims(axes = var_6230_axes_0, x = var_6228_cast_fp16)[name = string("op_6230_cast_fp16")]; + string query_states_81_pad_type_0 = const()[name = string("query_states_81_pad_type_0"), val = string("valid")]; + tensor query_states_81_strides_0 = const()[name = string("query_states_81_strides_0"), val = tensor([1, 1])]; + tensor query_states_81_pad_0 = const()[name = string("query_states_81_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_81_dilations_0 = const()[name = string("query_states_81_dilations_0"), val = tensor([1, 1])]; + int32 query_states_81_groups_0 = const()[name = string("query_states_81_groups_0"), val = int32(1)]; + tensor query_states_81 = conv(dilations = query_states_81_dilations_0, groups = query_states_81_groups_0, pad = query_states_81_pad_0, pad_type = query_states_81_pad_type_0, strides = query_states_81_strides_0, weight = model_model_layers_10_self_attn_q_proj_weight_palettized, x = var_6230_cast_fp16)[name = string("query_states_81")]; + string key_states_101_pad_type_0 = const()[name = string("key_states_101_pad_type_0"), val = string("valid")]; + tensor key_states_101_strides_0 = const()[name = string("key_states_101_strides_0"), val = tensor([1, 1])]; + tensor key_states_101_pad_0 = const()[name = string("key_states_101_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_101_dilations_0 = const()[name = string("key_states_101_dilations_0"), val = tensor([1, 1])]; + int32 key_states_101_groups_0 = const()[name = string("key_states_101_groups_0"), val = int32(1)]; + tensor key_states_101 = conv(dilations = key_states_101_dilations_0, groups = key_states_101_groups_0, pad = key_states_101_pad_0, pad_type = key_states_101_pad_type_0, strides = key_states_101_strides_0, weight = model_model_layers_10_self_attn_k_proj_weight_palettized, x = var_6230_cast_fp16)[name = string("key_states_101")]; + string value_states_81_pad_type_0 = const()[name = string("value_states_81_pad_type_0"), val = string("valid")]; + tensor value_states_81_strides_0 = const()[name = string("value_states_81_strides_0"), val = tensor([1, 1])]; + tensor value_states_81_pad_0 = const()[name = string("value_states_81_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_81_dilations_0 = const()[name = string("value_states_81_dilations_0"), val = tensor([1, 1])]; + int32 value_states_81_groups_0 = const()[name = string("value_states_81_groups_0"), val = int32(1)]; + tensor value_states_81 = conv(dilations = value_states_81_dilations_0, groups = value_states_81_groups_0, pad = value_states_81_pad_0, pad_type = value_states_81_pad_type_0, strides = value_states_81_strides_0, weight = model_model_layers_10_self_attn_v_proj_weight_palettized, x = var_6230_cast_fp16)[name = string("value_states_81")]; + tensor var_6272 = const()[name = string("op_6272"), val = tensor([1, 16, 128, 128])]; + tensor var_6273 = reshape(shape = var_6272, x = query_states_81)[name = string("op_6273")]; + tensor var_6278 = const()[name = string("op_6278"), val = tensor([0, 1, 3, 2])]; + tensor var_6283 = const()[name = string("op_6283"), val = tensor([1, 8, 128, 128])]; + tensor var_6284 = reshape(shape = var_6283, x = key_states_101)[name = string("op_6284")]; + tensor var_6289 = const()[name = string("op_6289"), val = tensor([0, 1, 3, 2])]; + tensor var_6294 = const()[name = string("op_6294"), val = tensor([1, 8, 128, 128])]; + tensor var_6295 = reshape(shape = var_6294, x = value_states_81)[name = string("op_6295")]; + tensor var_6300 = const()[name = string("op_6300"), val = tensor([0, 1, 3, 2])]; + int32 var_6311 = const()[name = string("op_6311"), val = int32(-1)]; + fp16 const_346_promoted = const()[name = string("const_346_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_105 = transpose(perm = var_6278, x = var_6273)[name = string("transpose_34")]; + tensor var_6313 = mul(x = hidden_states_105, y = const_346_promoted)[name = string("op_6313")]; + bool input_185_interleave_0 = const()[name = string("input_185_interleave_0"), val = bool(false)]; + tensor input_185 = concat(axis = var_6311, interleave = input_185_interleave_0, values = (hidden_states_105, var_6313))[name = string("input_185")]; + tensor normed_165_axes_0 = const()[name = string("normed_165_axes_0"), val = tensor([-1])]; + fp16 var_6308_to_fp16 = const()[name = string("op_6308_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_165_cast_fp16 = layer_norm(axes = normed_165_axes_0, epsilon = var_6308_to_fp16, x = input_185)[name = string("normed_165_cast_fp16")]; + tensor normed_167_begin_0 = const()[name = string("normed_167_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_167_end_0 = const()[name = string("normed_167_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_167_end_mask_0 = const()[name = string("normed_167_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_167 = slice_by_index(begin = normed_167_begin_0, end = normed_167_end_0, end_mask = normed_167_end_mask_0, x = normed_165_cast_fp16)[name = string("normed_167")]; + tensor const_349 = const()[name = string("const_349"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706846208)))]; + tensor q_21 = mul(x = normed_167, y = const_349)[name = string("q_21")]; + int32 var_6336 = const()[name = string("op_6336"), val = int32(-1)]; + fp16 const_350_promoted = const()[name = string("const_350_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_107 = transpose(perm = var_6289, x = var_6284)[name = string("transpose_33")]; + tensor var_6338 = mul(x = hidden_states_107, y = const_350_promoted)[name = string("op_6338")]; + bool input_187_interleave_0 = const()[name = string("input_187_interleave_0"), val = bool(false)]; + tensor input_187 = concat(axis = var_6336, interleave = input_187_interleave_0, values = (hidden_states_107, var_6338))[name = string("input_187")]; + tensor normed_169_axes_0 = const()[name = string("normed_169_axes_0"), val = tensor([-1])]; + fp16 var_6333_to_fp16 = const()[name = string("op_6333_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_169_cast_fp16 = layer_norm(axes = normed_169_axes_0, epsilon = var_6333_to_fp16, x = input_187)[name = string("normed_169_cast_fp16")]; + tensor normed_171_begin_0 = const()[name = string("normed_171_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_171_end_0 = const()[name = string("normed_171_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_171_end_mask_0 = const()[name = string("normed_171_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_171 = slice_by_index(begin = normed_171_begin_0, end = normed_171_end_0, end_mask = normed_171_end_mask_0, x = normed_169_cast_fp16)[name = string("normed_171")]; + tensor const_353 = const()[name = string("const_353"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706846528)))]; + tensor k_21 = mul(x = normed_171, y = const_353)[name = string("k_21")]; + tensor var_6364 = mul(x = q_21, y = cos_5)[name = string("op_6364")]; + tensor x1_41_begin_0 = const()[name = string("x1_41_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_41_end_0 = const()[name = string("x1_41_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_41_end_mask_0 = const()[name = string("x1_41_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_41 = slice_by_index(begin = x1_41_begin_0, end = x1_41_end_0, end_mask = x1_41_end_mask_0, x = q_21)[name = string("x1_41")]; + tensor x2_41_begin_0 = const()[name = string("x2_41_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_41_end_0 = const()[name = string("x2_41_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_41_end_mask_0 = const()[name = string("x2_41_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_41 = slice_by_index(begin = x2_41_begin_0, end = x2_41_end_0, end_mask = x2_41_end_mask_0, x = q_21)[name = string("x2_41")]; + fp16 const_356_promoted = const()[name = string("const_356_promoted"), val = fp16(-0x1p+0)]; + tensor var_6385 = mul(x = x2_41, y = const_356_promoted)[name = string("op_6385")]; + int32 var_6387 = const()[name = string("op_6387"), val = int32(-1)]; + bool var_6388_interleave_0 = const()[name = string("op_6388_interleave_0"), val = bool(false)]; + tensor var_6388 = concat(axis = var_6387, interleave = var_6388_interleave_0, values = (var_6385, x1_41))[name = string("op_6388")]; + tensor var_6389 = mul(x = var_6388, y = sin_5)[name = string("op_6389")]; + tensor query_states_83 = add(x = var_6364, y = var_6389)[name = string("query_states_83")]; + tensor var_6392 = mul(x = k_21, y = cos_5)[name = string("op_6392")]; + tensor x1_43_begin_0 = const()[name = string("x1_43_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_43_end_0 = const()[name = string("x1_43_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_43_end_mask_0 = const()[name = string("x1_43_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_43 = slice_by_index(begin = x1_43_begin_0, end = x1_43_end_0, end_mask = x1_43_end_mask_0, x = k_21)[name = string("x1_43")]; + tensor x2_43_begin_0 = const()[name = string("x2_43_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_43_end_0 = const()[name = string("x2_43_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_43_end_mask_0 = const()[name = string("x2_43_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_43 = slice_by_index(begin = x2_43_begin_0, end = x2_43_end_0, end_mask = x2_43_end_mask_0, x = k_21)[name = string("x2_43")]; + fp16 const_359_promoted = const()[name = string("const_359_promoted"), val = fp16(-0x1p+0)]; + tensor var_6413 = mul(x = x2_43, y = const_359_promoted)[name = string("op_6413")]; + int32 var_6415 = const()[name = string("op_6415"), val = int32(-1)]; + bool var_6416_interleave_0 = const()[name = string("op_6416_interleave_0"), val = bool(false)]; + tensor var_6416 = concat(axis = var_6415, interleave = var_6416_interleave_0, values = (var_6413, x1_43))[name = string("op_6416")]; + tensor var_6417 = mul(x = var_6416, y = sin_5)[name = string("op_6417")]; + tensor key_states_103 = add(x = var_6392, y = var_6417)[name = string("key_states_103")]; + tensor expand_dims_120 = const()[name = string("expand_dims_120"), val = tensor([10])]; + tensor expand_dims_121 = const()[name = string("expand_dims_121"), val = tensor([0])]; + tensor expand_dims_123 = const()[name = string("expand_dims_123"), val = tensor([0])]; + tensor expand_dims_124 = const()[name = string("expand_dims_124"), val = tensor([11])]; + int32 concat_182_axis_0 = const()[name = string("concat_182_axis_0"), val = int32(0)]; + bool concat_182_interleave_0 = const()[name = string("concat_182_interleave_0"), val = bool(false)]; + tensor concat_182 = concat(axis = concat_182_axis_0, interleave = concat_182_interleave_0, values = (expand_dims_120, expand_dims_121, current_pos, expand_dims_123))[name = string("concat_182")]; + tensor concat_183_values1_0 = const()[name = string("concat_183_values1_0"), val = tensor([0])]; + tensor concat_183_values3_0 = const()[name = string("concat_183_values3_0"), val = tensor([0])]; + int32 concat_183_axis_0 = const()[name = string("concat_183_axis_0"), val = int32(0)]; + bool concat_183_interleave_0 = const()[name = string("concat_183_interleave_0"), val = bool(false)]; + tensor concat_183 = concat(axis = concat_183_axis_0, interleave = concat_183_interleave_0, values = (expand_dims_124, concat_183_values1_0, var_1039, concat_183_values3_0))[name = string("concat_183")]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_182, begin_mask = model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0, end = concat_183, end_mask = model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_21_stride_0, update = key_states_103, x = coreml_update_state_47)[name = string("model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_104_write_state")]; + tensor coreml_update_state_48 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_104")]; + tensor expand_dims_126 = const()[name = string("expand_dims_126"), val = tensor([38])]; + tensor expand_dims_127 = const()[name = string("expand_dims_127"), val = tensor([0])]; + tensor expand_dims_129 = const()[name = string("expand_dims_129"), val = tensor([0])]; + tensor expand_dims_130 = const()[name = string("expand_dims_130"), val = tensor([39])]; + int32 concat_186_axis_0 = const()[name = string("concat_186_axis_0"), val = int32(0)]; + bool concat_186_interleave_0 = const()[name = string("concat_186_interleave_0"), val = bool(false)]; + tensor concat_186 = concat(axis = concat_186_axis_0, interleave = concat_186_interleave_0, values = (expand_dims_126, expand_dims_127, current_pos, expand_dims_129))[name = string("concat_186")]; + tensor concat_187_values1_0 = const()[name = string("concat_187_values1_0"), val = tensor([0])]; + tensor concat_187_values3_0 = const()[name = string("concat_187_values3_0"), val = tensor([0])]; + int32 concat_187_axis_0 = const()[name = string("concat_187_axis_0"), val = int32(0)]; + bool concat_187_interleave_0 = const()[name = string("concat_187_interleave_0"), val = bool(false)]; + tensor concat_187 = concat(axis = concat_187_axis_0, interleave = concat_187_interleave_0, values = (expand_dims_130, concat_187_values1_0, var_1039, concat_187_values3_0))[name = string("concat_187")]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_83 = transpose(perm = var_6300, x = var_6295)[name = string("transpose_32")]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_186, begin_mask = model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0, end = concat_187, end_mask = model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_22_stride_0, update = value_states_83, x = coreml_update_state_48)[name = string("model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_105_write_state")]; + tensor coreml_update_state_49 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_105")]; + tensor var_6488_begin_0 = const()[name = string("op_6488_begin_0"), val = tensor([10, 0, 0, 0])]; + tensor var_6488_end_0 = const()[name = string("op_6488_end_0"), val = tensor([11, 8, 1024, 128])]; + tensor var_6488_end_mask_0 = const()[name = string("op_6488_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_6488_cast_fp16 = slice_by_index(begin = var_6488_begin_0, end = var_6488_end_0, end_mask = var_6488_end_mask_0, x = coreml_update_state_49)[name = string("op_6488_cast_fp16")]; + tensor K_layer_cache_21_axes_0 = const()[name = string("K_layer_cache_21_axes_0"), val = tensor([0])]; + tensor K_layer_cache_21_cast_fp16 = squeeze(axes = K_layer_cache_21_axes_0, x = var_6488_cast_fp16)[name = string("K_layer_cache_21_cast_fp16")]; + tensor var_6495_begin_0 = const()[name = string("op_6495_begin_0"), val = tensor([38, 0, 0, 0])]; + tensor var_6495_end_0 = const()[name = string("op_6495_end_0"), val = tensor([39, 8, 1024, 128])]; + tensor var_6495_end_mask_0 = const()[name = string("op_6495_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_6495_cast_fp16 = slice_by_index(begin = var_6495_begin_0, end = var_6495_end_0, end_mask = var_6495_end_mask_0, x = coreml_update_state_49)[name = string("op_6495_cast_fp16")]; + tensor V_layer_cache_21_axes_0 = const()[name = string("V_layer_cache_21_axes_0"), val = tensor([0])]; + tensor V_layer_cache_21_cast_fp16 = squeeze(axes = V_layer_cache_21_axes_0, x = var_6495_cast_fp16)[name = string("V_layer_cache_21_cast_fp16")]; + tensor x_163_axes_0 = const()[name = string("x_163_axes_0"), val = tensor([1])]; + tensor x_163_cast_fp16 = expand_dims(axes = x_163_axes_0, x = K_layer_cache_21_cast_fp16)[name = string("x_163_cast_fp16")]; + tensor var_6524 = const()[name = string("op_6524"), val = tensor([1, 2, 1, 1])]; + tensor x_165_cast_fp16 = tile(reps = var_6524, x = x_163_cast_fp16)[name = string("x_165_cast_fp16")]; + tensor var_6536 = const()[name = string("op_6536"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_107_cast_fp16 = reshape(shape = var_6536, x = x_165_cast_fp16)[name = string("key_states_107_cast_fp16")]; + tensor x_169_axes_0 = const()[name = string("x_169_axes_0"), val = tensor([1])]; + tensor x_169_cast_fp16 = expand_dims(axes = x_169_axes_0, x = V_layer_cache_21_cast_fp16)[name = string("x_169_cast_fp16")]; + tensor var_6544 = const()[name = string("op_6544"), val = tensor([1, 2, 1, 1])]; + tensor x_171_cast_fp16 = tile(reps = var_6544, x = x_169_cast_fp16)[name = string("x_171_cast_fp16")]; + bool var_6571_transpose_x_0 = const()[name = string("op_6571_transpose_x_0"), val = bool(false)]; + bool var_6571_transpose_y_0 = const()[name = string("op_6571_transpose_y_0"), val = bool(true)]; + tensor var_6571 = matmul(transpose_x = var_6571_transpose_x_0, transpose_y = var_6571_transpose_y_0, x = query_states_83, y = key_states_107_cast_fp16)[name = string("op_6571")]; + fp16 var_6572_to_fp16 = const()[name = string("op_6572_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_41_cast_fp16 = mul(x = var_6571, y = var_6572_to_fp16)[name = string("attn_weights_41_cast_fp16")]; + tensor attn_weights_43_cast_fp16 = add(x = attn_weights_41_cast_fp16, y = causal_mask)[name = string("attn_weights_43_cast_fp16")]; + int32 var_6607 = const()[name = string("op_6607"), val = int32(-1)]; + tensor var_6609_cast_fp16 = softmax(axis = var_6607, x = attn_weights_43_cast_fp16)[name = string("op_6609_cast_fp16")]; + tensor concat_192 = const()[name = string("concat_192"), val = tensor([16, 128, 1024])]; + tensor reshape_30_cast_fp16 = reshape(shape = concat_192, x = var_6609_cast_fp16)[name = string("reshape_30_cast_fp16")]; + tensor concat_193 = const()[name = string("concat_193"), val = tensor([16, 1024, 128])]; + tensor reshape_31_cast_fp16 = reshape(shape = concat_193, x = x_171_cast_fp16)[name = string("reshape_31_cast_fp16")]; + bool matmul_10_transpose_x_0 = const()[name = string("matmul_10_transpose_x_0"), val = bool(false)]; + bool matmul_10_transpose_y_0 = const()[name = string("matmul_10_transpose_y_0"), val = bool(false)]; + tensor matmul_10_cast_fp16 = matmul(transpose_x = matmul_10_transpose_x_0, transpose_y = matmul_10_transpose_y_0, x = reshape_30_cast_fp16, y = reshape_31_cast_fp16)[name = string("matmul_10_cast_fp16")]; + tensor concat_197 = const()[name = string("concat_197"), val = tensor([1, 16, 128, 128])]; + tensor reshape_32_cast_fp16 = reshape(shape = concat_197, x = matmul_10_cast_fp16)[name = string("reshape_32_cast_fp16")]; + tensor var_6621_perm_0 = const()[name = string("op_6621_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_6640 = const()[name = string("op_6640"), val = tensor([1, 128, 2048])]; + tensor var_6621_cast_fp16 = transpose(perm = var_6621_perm_0, x = reshape_32_cast_fp16)[name = string("transpose_31")]; + tensor attn_output_105_cast_fp16 = reshape(shape = var_6640, x = var_6621_cast_fp16)[name = string("attn_output_105_cast_fp16")]; + tensor var_6645 = const()[name = string("op_6645"), val = tensor([0, 2, 1])]; + string var_6661_pad_type_0 = const()[name = string("op_6661_pad_type_0"), val = string("valid")]; + int32 var_6661_groups_0 = const()[name = string("op_6661_groups_0"), val = int32(1)]; + tensor var_6661_strides_0 = const()[name = string("op_6661_strides_0"), val = tensor([1])]; + tensor var_6661_pad_0 = const()[name = string("op_6661_pad_0"), val = tensor([0, 0])]; + tensor var_6661_dilations_0 = const()[name = string("op_6661_dilations_0"), val = tensor([1])]; + tensor squeeze_10_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706846848))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711041216))))[name = string("squeeze_10_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_6646_cast_fp16 = transpose(perm = var_6645, x = attn_output_105_cast_fp16)[name = string("transpose_30")]; + tensor var_6661_cast_fp16 = conv(dilations = var_6661_dilations_0, groups = var_6661_groups_0, pad = var_6661_pad_0, pad_type = var_6661_pad_type_0, strides = var_6661_strides_0, weight = squeeze_10_cast_fp16_to_fp32_to_fp16_palettized, x = var_6646_cast_fp16)[name = string("op_6661_cast_fp16")]; + tensor var_6665 = const()[name = string("op_6665"), val = tensor([0, 2, 1])]; + tensor attn_output_109_cast_fp16 = transpose(perm = var_6665, x = var_6661_cast_fp16)[name = string("transpose_29")]; + tensor hidden_states_109_cast_fp16 = add(x = hidden_states_101_cast_fp16, y = attn_output_109_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; + int32 var_6678 = const()[name = string("op_6678"), val = int32(-1)]; + fp16 const_371_promoted_to_fp16 = const()[name = string("const_371_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6680_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = const_371_promoted_to_fp16)[name = string("op_6680_cast_fp16")]; + bool input_191_interleave_0 = const()[name = string("input_191_interleave_0"), val = bool(false)]; + tensor input_191_cast_fp16 = concat(axis = var_6678, interleave = input_191_interleave_0, values = (hidden_states_109_cast_fp16, var_6680_cast_fp16))[name = string("input_191_cast_fp16")]; + tensor normed_173_axes_0 = const()[name = string("normed_173_axes_0"), val = tensor([-1])]; + fp16 var_6675_to_fp16 = const()[name = string("op_6675_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_173_cast_fp16 = layer_norm(axes = normed_173_axes_0, epsilon = var_6675_to_fp16, x = input_191_cast_fp16)[name = string("normed_173_cast_fp16")]; + tensor normed_175_begin_0 = const()[name = string("normed_175_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_175_end_0 = const()[name = string("normed_175_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_175_end_mask_0 = const()[name = string("normed_175_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_175_cast_fp16 = slice_by_index(begin = normed_175_begin_0, end = normed_175_end_0, end_mask = normed_175_end_mask_0, x = normed_173_cast_fp16)[name = string("normed_175_cast_fp16")]; + tensor const_374_promoted_to_fp16 = const()[name = string("const_374_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711172352)))]; + tensor x_173_cast_fp16 = mul(x = normed_175_cast_fp16, y = const_374_promoted_to_fp16)[name = string("x_173_cast_fp16")]; + tensor var_6705 = const()[name = string("op_6705"), val = tensor([0, 2, 1])]; + tensor input_193_axes_0 = const()[name = string("input_193_axes_0"), val = tensor([2])]; + tensor var_6706 = transpose(perm = var_6705, x = x_173_cast_fp16)[name = string("transpose_28")]; + tensor input_193 = expand_dims(axes = input_193_axes_0, x = var_6706)[name = string("input_193")]; + string input_195_pad_type_0 = const()[name = string("input_195_pad_type_0"), val = string("valid")]; + tensor input_195_strides_0 = const()[name = string("input_195_strides_0"), val = tensor([1, 1])]; + tensor input_195_pad_0 = const()[name = string("input_195_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_195_dilations_0 = const()[name = string("input_195_dilations_0"), val = tensor([1, 1])]; + int32 input_195_groups_0 = const()[name = string("input_195_groups_0"), val = int32(1)]; + tensor input_195 = conv(dilations = input_195_dilations_0, groups = input_195_groups_0, pad = input_195_pad_0, pad_type = input_195_pad_type_0, strides = input_195_strides_0, weight = model_model_layers_10_mlp_gate_proj_weight_palettized, x = input_193)[name = string("input_195")]; + string b_21_pad_type_0 = const()[name = string("b_21_pad_type_0"), val = string("valid")]; + tensor b_21_strides_0 = const()[name = string("b_21_strides_0"), val = tensor([1, 1])]; + tensor b_21_pad_0 = const()[name = string("b_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_21_dilations_0 = const()[name = string("b_21_dilations_0"), val = tensor([1, 1])]; + int32 b_21_groups_0 = const()[name = string("b_21_groups_0"), val = int32(1)]; + tensor b_21 = conv(dilations = b_21_dilations_0, groups = b_21_groups_0, pad = b_21_pad_0, pad_type = b_21_pad_type_0, strides = b_21_strides_0, weight = model_model_layers_10_mlp_up_proj_weight_palettized, x = input_193)[name = string("b_21")]; + tensor c_21 = silu(x = input_195)[name = string("c_21")]; + tensor input_197 = mul(x = c_21, y = b_21)[name = string("input_197")]; + string e_21_pad_type_0 = const()[name = string("e_21_pad_type_0"), val = string("valid")]; + tensor e_21_strides_0 = const()[name = string("e_21_strides_0"), val = tensor([1, 1])]; + tensor e_21_pad_0 = const()[name = string("e_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_21_dilations_0 = const()[name = string("e_21_dilations_0"), val = tensor([1, 1])]; + int32 e_21_groups_0 = const()[name = string("e_21_groups_0"), val = int32(1)]; + tensor e_21 = conv(dilations = e_21_dilations_0, groups = e_21_groups_0, pad = e_21_pad_0, pad_type = e_21_pad_type_0, strides = e_21_strides_0, weight = model_model_layers_10_mlp_down_proj_weight_palettized, x = input_197)[name = string("e_21")]; + tensor var_6728_axes_0 = const()[name = string("op_6728_axes_0"), val = tensor([2])]; + tensor var_6728 = squeeze(axes = var_6728_axes_0, x = e_21)[name = string("op_6728")]; + tensor var_6729 = const()[name = string("op_6729"), val = tensor([0, 2, 1])]; + tensor var_6730 = transpose(perm = var_6729, x = var_6728)[name = string("transpose_27")]; + tensor hidden_states_111_cast_fp16 = add(x = hidden_states_109_cast_fp16, y = var_6730)[name = string("hidden_states_111_cast_fp16")]; + int32 var_6742 = const()[name = string("op_6742"), val = int32(-1)]; + fp16 const_375_promoted_to_fp16 = const()[name = string("const_375_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6744_cast_fp16 = mul(x = hidden_states_111_cast_fp16, y = const_375_promoted_to_fp16)[name = string("op_6744_cast_fp16")]; + bool input_199_interleave_0 = const()[name = string("input_199_interleave_0"), val = bool(false)]; + tensor input_199_cast_fp16 = concat(axis = var_6742, interleave = input_199_interleave_0, values = (hidden_states_111_cast_fp16, var_6744_cast_fp16))[name = string("input_199_cast_fp16")]; + tensor normed_177_axes_0 = const()[name = string("normed_177_axes_0"), val = tensor([-1])]; + fp16 var_6739_to_fp16 = const()[name = string("op_6739_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_177_cast_fp16 = layer_norm(axes = normed_177_axes_0, epsilon = var_6739_to_fp16, x = input_199_cast_fp16)[name = string("normed_177_cast_fp16")]; + tensor normed_179_begin_0 = const()[name = string("normed_179_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_179_end_0 = const()[name = string("normed_179_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_179_end_mask_0 = const()[name = string("normed_179_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_179_cast_fp16 = slice_by_index(begin = normed_179_begin_0, end = normed_179_end_0, end_mask = normed_179_end_mask_0, x = normed_177_cast_fp16)[name = string("normed_179_cast_fp16")]; + tensor const_378_promoted_to_fp16 = const()[name = string("const_378_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711176512)))]; + tensor hidden_states_113_cast_fp16 = mul(x = normed_179_cast_fp16, y = const_378_promoted_to_fp16)[name = string("hidden_states_113_cast_fp16")]; + tensor var_6767 = const()[name = string("op_6767"), val = tensor([0, 2, 1])]; + tensor var_6770_axes_0 = const()[name = string("op_6770_axes_0"), val = tensor([2])]; + tensor var_6768_cast_fp16 = transpose(perm = var_6767, x = hidden_states_113_cast_fp16)[name = string("transpose_26")]; + tensor var_6770_cast_fp16 = expand_dims(axes = var_6770_axes_0, x = var_6768_cast_fp16)[name = string("op_6770_cast_fp16")]; + string query_states_89_pad_type_0 = const()[name = string("query_states_89_pad_type_0"), val = string("valid")]; + tensor query_states_89_strides_0 = const()[name = string("query_states_89_strides_0"), val = tensor([1, 1])]; + tensor query_states_89_pad_0 = const()[name = string("query_states_89_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_89_dilations_0 = const()[name = string("query_states_89_dilations_0"), val = tensor([1, 1])]; + int32 query_states_89_groups_0 = const()[name = string("query_states_89_groups_0"), val = int32(1)]; + tensor query_states_89 = conv(dilations = query_states_89_dilations_0, groups = query_states_89_groups_0, pad = query_states_89_pad_0, pad_type = query_states_89_pad_type_0, strides = query_states_89_strides_0, weight = model_model_layers_11_self_attn_q_proj_weight_palettized, x = var_6770_cast_fp16)[name = string("query_states_89")]; + string key_states_111_pad_type_0 = const()[name = string("key_states_111_pad_type_0"), val = string("valid")]; + tensor key_states_111_strides_0 = const()[name = string("key_states_111_strides_0"), val = tensor([1, 1])]; + tensor key_states_111_pad_0 = const()[name = string("key_states_111_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_111_dilations_0 = const()[name = string("key_states_111_dilations_0"), val = tensor([1, 1])]; + int32 key_states_111_groups_0 = const()[name = string("key_states_111_groups_0"), val = int32(1)]; + tensor key_states_111 = conv(dilations = key_states_111_dilations_0, groups = key_states_111_groups_0, pad = key_states_111_pad_0, pad_type = key_states_111_pad_type_0, strides = key_states_111_strides_0, weight = model_model_layers_11_self_attn_k_proj_weight_palettized, x = var_6770_cast_fp16)[name = string("key_states_111")]; + string value_states_89_pad_type_0 = const()[name = string("value_states_89_pad_type_0"), val = string("valid")]; + tensor value_states_89_strides_0 = const()[name = string("value_states_89_strides_0"), val = tensor([1, 1])]; + tensor value_states_89_pad_0 = const()[name = string("value_states_89_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_89_dilations_0 = const()[name = string("value_states_89_dilations_0"), val = tensor([1, 1])]; + int32 value_states_89_groups_0 = const()[name = string("value_states_89_groups_0"), val = int32(1)]; + tensor value_states_89 = conv(dilations = value_states_89_dilations_0, groups = value_states_89_groups_0, pad = value_states_89_pad_0, pad_type = value_states_89_pad_type_0, strides = value_states_89_strides_0, weight = model_model_layers_11_self_attn_v_proj_weight_palettized, x = var_6770_cast_fp16)[name = string("value_states_89")]; + tensor var_6812 = const()[name = string("op_6812"), val = tensor([1, 16, 128, 128])]; + tensor var_6813 = reshape(shape = var_6812, x = query_states_89)[name = string("op_6813")]; + tensor var_6818 = const()[name = string("op_6818"), val = tensor([0, 1, 3, 2])]; + tensor var_6823 = const()[name = string("op_6823"), val = tensor([1, 8, 128, 128])]; + tensor var_6824 = reshape(shape = var_6823, x = key_states_111)[name = string("op_6824")]; + tensor var_6829 = const()[name = string("op_6829"), val = tensor([0, 1, 3, 2])]; + tensor var_6834 = const()[name = string("op_6834"), val = tensor([1, 8, 128, 128])]; + tensor var_6835 = reshape(shape = var_6834, x = value_states_89)[name = string("op_6835")]; + tensor var_6840 = const()[name = string("op_6840"), val = tensor([0, 1, 3, 2])]; + int32 var_6851 = const()[name = string("op_6851"), val = int32(-1)]; + fp16 const_380_promoted = const()[name = string("const_380_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_115 = transpose(perm = var_6818, x = var_6813)[name = string("transpose_25")]; + tensor var_6853 = mul(x = hidden_states_115, y = const_380_promoted)[name = string("op_6853")]; + bool input_203_interleave_0 = const()[name = string("input_203_interleave_0"), val = bool(false)]; + tensor input_203 = concat(axis = var_6851, interleave = input_203_interleave_0, values = (hidden_states_115, var_6853))[name = string("input_203")]; + tensor normed_181_axes_0 = const()[name = string("normed_181_axes_0"), val = tensor([-1])]; + fp16 var_6848_to_fp16 = const()[name = string("op_6848_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_181_cast_fp16 = layer_norm(axes = normed_181_axes_0, epsilon = var_6848_to_fp16, x = input_203)[name = string("normed_181_cast_fp16")]; + tensor normed_183_begin_0 = const()[name = string("normed_183_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_183_end_0 = const()[name = string("normed_183_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_183_end_mask_0 = const()[name = string("normed_183_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_183 = slice_by_index(begin = normed_183_begin_0, end = normed_183_end_0, end_mask = normed_183_end_mask_0, x = normed_181_cast_fp16)[name = string("normed_183")]; + tensor const_383 = const()[name = string("const_383"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711180672)))]; + tensor q_23 = mul(x = normed_183, y = const_383)[name = string("q_23")]; + int32 var_6876 = const()[name = string("op_6876"), val = int32(-1)]; + fp16 const_384_promoted = const()[name = string("const_384_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_117 = transpose(perm = var_6829, x = var_6824)[name = string("transpose_24")]; + tensor var_6878 = mul(x = hidden_states_117, y = const_384_promoted)[name = string("op_6878")]; + bool input_205_interleave_0 = const()[name = string("input_205_interleave_0"), val = bool(false)]; + tensor input_205 = concat(axis = var_6876, interleave = input_205_interleave_0, values = (hidden_states_117, var_6878))[name = string("input_205")]; + tensor normed_185_axes_0 = const()[name = string("normed_185_axes_0"), val = tensor([-1])]; + fp16 var_6873_to_fp16 = const()[name = string("op_6873_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_185_cast_fp16 = layer_norm(axes = normed_185_axes_0, epsilon = var_6873_to_fp16, x = input_205)[name = string("normed_185_cast_fp16")]; + tensor normed_187_begin_0 = const()[name = string("normed_187_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_187_end_0 = const()[name = string("normed_187_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_187_end_mask_0 = const()[name = string("normed_187_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_187 = slice_by_index(begin = normed_187_begin_0, end = normed_187_end_0, end_mask = normed_187_end_mask_0, x = normed_185_cast_fp16)[name = string("normed_187")]; + tensor const_387 = const()[name = string("const_387"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711180992)))]; + tensor k_23 = mul(x = normed_187, y = const_387)[name = string("k_23")]; + tensor var_6904 = mul(x = q_23, y = cos_5)[name = string("op_6904")]; + tensor x1_45_begin_0 = const()[name = string("x1_45_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_45_end_0 = const()[name = string("x1_45_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_45_end_mask_0 = const()[name = string("x1_45_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_45 = slice_by_index(begin = x1_45_begin_0, end = x1_45_end_0, end_mask = x1_45_end_mask_0, x = q_23)[name = string("x1_45")]; + tensor x2_45_begin_0 = const()[name = string("x2_45_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_45_end_0 = const()[name = string("x2_45_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_45_end_mask_0 = const()[name = string("x2_45_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_45 = slice_by_index(begin = x2_45_begin_0, end = x2_45_end_0, end_mask = x2_45_end_mask_0, x = q_23)[name = string("x2_45")]; + fp16 const_390_promoted = const()[name = string("const_390_promoted"), val = fp16(-0x1p+0)]; + tensor var_6925 = mul(x = x2_45, y = const_390_promoted)[name = string("op_6925")]; + int32 var_6927 = const()[name = string("op_6927"), val = int32(-1)]; + bool var_6928_interleave_0 = const()[name = string("op_6928_interleave_0"), val = bool(false)]; + tensor var_6928 = concat(axis = var_6927, interleave = var_6928_interleave_0, values = (var_6925, x1_45))[name = string("op_6928")]; + tensor var_6929 = mul(x = var_6928, y = sin_5)[name = string("op_6929")]; + tensor query_states_91 = add(x = var_6904, y = var_6929)[name = string("query_states_91")]; + tensor var_6932 = mul(x = k_23, y = cos_5)[name = string("op_6932")]; + tensor x1_47_begin_0 = const()[name = string("x1_47_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_47_end_0 = const()[name = string("x1_47_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_47_end_mask_0 = const()[name = string("x1_47_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_47 = slice_by_index(begin = x1_47_begin_0, end = x1_47_end_0, end_mask = x1_47_end_mask_0, x = k_23)[name = string("x1_47")]; + tensor x2_47_begin_0 = const()[name = string("x2_47_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_47_end_0 = const()[name = string("x2_47_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_47_end_mask_0 = const()[name = string("x2_47_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_47 = slice_by_index(begin = x2_47_begin_0, end = x2_47_end_0, end_mask = x2_47_end_mask_0, x = k_23)[name = string("x2_47")]; + fp16 const_393_promoted = const()[name = string("const_393_promoted"), val = fp16(-0x1p+0)]; + tensor var_6953 = mul(x = x2_47, y = const_393_promoted)[name = string("op_6953")]; + int32 var_6955 = const()[name = string("op_6955"), val = int32(-1)]; + bool var_6956_interleave_0 = const()[name = string("op_6956_interleave_0"), val = bool(false)]; + tensor var_6956 = concat(axis = var_6955, interleave = var_6956_interleave_0, values = (var_6953, x1_47))[name = string("op_6956")]; + tensor var_6957 = mul(x = var_6956, y = sin_5)[name = string("op_6957")]; + tensor key_states_113 = add(x = var_6932, y = var_6957)[name = string("key_states_113")]; + tensor expand_dims_132 = const()[name = string("expand_dims_132"), val = tensor([11])]; + tensor expand_dims_133 = const()[name = string("expand_dims_133"), val = tensor([0])]; + tensor expand_dims_135 = const()[name = string("expand_dims_135"), val = tensor([0])]; + tensor expand_dims_136 = const()[name = string("expand_dims_136"), val = tensor([12])]; + int32 concat_200_axis_0 = const()[name = string("concat_200_axis_0"), val = int32(0)]; + bool concat_200_interleave_0 = const()[name = string("concat_200_interleave_0"), val = bool(false)]; + tensor concat_200 = concat(axis = concat_200_axis_0, interleave = concat_200_interleave_0, values = (expand_dims_132, expand_dims_133, current_pos, expand_dims_135))[name = string("concat_200")]; + tensor concat_201_values1_0 = const()[name = string("concat_201_values1_0"), val = tensor([0])]; + tensor concat_201_values3_0 = const()[name = string("concat_201_values3_0"), val = tensor([0])]; + int32 concat_201_axis_0 = const()[name = string("concat_201_axis_0"), val = int32(0)]; + bool concat_201_interleave_0 = const()[name = string("concat_201_interleave_0"), val = bool(false)]; + tensor concat_201 = concat(axis = concat_201_axis_0, interleave = concat_201_interleave_0, values = (expand_dims_136, concat_201_values1_0, var_1039, concat_201_values3_0))[name = string("concat_201")]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_200, begin_mask = model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0, end = concat_201, end_mask = model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_23_stride_0, update = key_states_113, x = coreml_update_state_49)[name = string("model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_106_write_state")]; + tensor coreml_update_state_50 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_106")]; + tensor expand_dims_138 = const()[name = string("expand_dims_138"), val = tensor([39])]; + tensor expand_dims_139 = const()[name = string("expand_dims_139"), val = tensor([0])]; + tensor expand_dims_141 = const()[name = string("expand_dims_141"), val = tensor([0])]; + tensor expand_dims_142 = const()[name = string("expand_dims_142"), val = tensor([40])]; + int32 concat_204_axis_0 = const()[name = string("concat_204_axis_0"), val = int32(0)]; + bool concat_204_interleave_0 = const()[name = string("concat_204_interleave_0"), val = bool(false)]; + tensor concat_204 = concat(axis = concat_204_axis_0, interleave = concat_204_interleave_0, values = (expand_dims_138, expand_dims_139, current_pos, expand_dims_141))[name = string("concat_204")]; + tensor concat_205_values1_0 = const()[name = string("concat_205_values1_0"), val = tensor([0])]; + tensor concat_205_values3_0 = const()[name = string("concat_205_values3_0"), val = tensor([0])]; + int32 concat_205_axis_0 = const()[name = string("concat_205_axis_0"), val = int32(0)]; + bool concat_205_interleave_0 = const()[name = string("concat_205_interleave_0"), val = bool(false)]; + tensor concat_205 = concat(axis = concat_205_axis_0, interleave = concat_205_interleave_0, values = (expand_dims_142, concat_205_values1_0, var_1039, concat_205_values3_0))[name = string("concat_205")]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_91 = transpose(perm = var_6840, x = var_6835)[name = string("transpose_23")]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_204, begin_mask = model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0, end = concat_205, end_mask = model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_24_stride_0, update = value_states_91, x = coreml_update_state_50)[name = string("model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_107_write_state")]; + tensor coreml_update_state_51 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_107")]; + tensor var_7028_begin_0 = const()[name = string("op_7028_begin_0"), val = tensor([11, 0, 0, 0])]; + tensor var_7028_end_0 = const()[name = string("op_7028_end_0"), val = tensor([12, 8, 1024, 128])]; + tensor var_7028_end_mask_0 = const()[name = string("op_7028_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_7028_cast_fp16 = slice_by_index(begin = var_7028_begin_0, end = var_7028_end_0, end_mask = var_7028_end_mask_0, x = coreml_update_state_51)[name = string("op_7028_cast_fp16")]; + tensor K_layer_cache_23_axes_0 = const()[name = string("K_layer_cache_23_axes_0"), val = tensor([0])]; + tensor K_layer_cache_23_cast_fp16 = squeeze(axes = K_layer_cache_23_axes_0, x = var_7028_cast_fp16)[name = string("K_layer_cache_23_cast_fp16")]; + tensor var_7035_begin_0 = const()[name = string("op_7035_begin_0"), val = tensor([39, 0, 0, 0])]; + tensor var_7035_end_0 = const()[name = string("op_7035_end_0"), val = tensor([40, 8, 1024, 128])]; + tensor var_7035_end_mask_0 = const()[name = string("op_7035_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_7035_cast_fp16 = slice_by_index(begin = var_7035_begin_0, end = var_7035_end_0, end_mask = var_7035_end_mask_0, x = coreml_update_state_51)[name = string("op_7035_cast_fp16")]; + tensor V_layer_cache_23_axes_0 = const()[name = string("V_layer_cache_23_axes_0"), val = tensor([0])]; + tensor V_layer_cache_23_cast_fp16 = squeeze(axes = V_layer_cache_23_axes_0, x = var_7035_cast_fp16)[name = string("V_layer_cache_23_cast_fp16")]; + tensor x_179_axes_0 = const()[name = string("x_179_axes_0"), val = tensor([1])]; + tensor x_179_cast_fp16 = expand_dims(axes = x_179_axes_0, x = K_layer_cache_23_cast_fp16)[name = string("x_179_cast_fp16")]; + tensor var_7064 = const()[name = string("op_7064"), val = tensor([1, 2, 1, 1])]; + tensor x_181_cast_fp16 = tile(reps = var_7064, x = x_179_cast_fp16)[name = string("x_181_cast_fp16")]; + tensor var_7076 = const()[name = string("op_7076"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_117_cast_fp16 = reshape(shape = var_7076, x = x_181_cast_fp16)[name = string("key_states_117_cast_fp16")]; + tensor x_185_axes_0 = const()[name = string("x_185_axes_0"), val = tensor([1])]; + tensor x_185_cast_fp16 = expand_dims(axes = x_185_axes_0, x = V_layer_cache_23_cast_fp16)[name = string("x_185_cast_fp16")]; + tensor var_7084 = const()[name = string("op_7084"), val = tensor([1, 2, 1, 1])]; + tensor x_187_cast_fp16 = tile(reps = var_7084, x = x_185_cast_fp16)[name = string("x_187_cast_fp16")]; + bool var_7111_transpose_x_0 = const()[name = string("op_7111_transpose_x_0"), val = bool(false)]; + bool var_7111_transpose_y_0 = const()[name = string("op_7111_transpose_y_0"), val = bool(true)]; + tensor var_7111 = matmul(transpose_x = var_7111_transpose_x_0, transpose_y = var_7111_transpose_y_0, x = query_states_91, y = key_states_117_cast_fp16)[name = string("op_7111")]; + fp16 var_7112_to_fp16 = const()[name = string("op_7112_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_45_cast_fp16 = mul(x = var_7111, y = var_7112_to_fp16)[name = string("attn_weights_45_cast_fp16")]; + tensor attn_weights_47_cast_fp16 = add(x = attn_weights_45_cast_fp16, y = causal_mask)[name = string("attn_weights_47_cast_fp16")]; + int32 var_7147 = const()[name = string("op_7147"), val = int32(-1)]; + tensor var_7149_cast_fp16 = softmax(axis = var_7147, x = attn_weights_47_cast_fp16)[name = string("op_7149_cast_fp16")]; + tensor concat_210 = const()[name = string("concat_210"), val = tensor([16, 128, 1024])]; + tensor reshape_33_cast_fp16 = reshape(shape = concat_210, x = var_7149_cast_fp16)[name = string("reshape_33_cast_fp16")]; + tensor concat_211 = const()[name = string("concat_211"), val = tensor([16, 1024, 128])]; + tensor reshape_34_cast_fp16 = reshape(shape = concat_211, x = x_187_cast_fp16)[name = string("reshape_34_cast_fp16")]; + bool matmul_11_transpose_x_0 = const()[name = string("matmul_11_transpose_x_0"), val = bool(false)]; + bool matmul_11_transpose_y_0 = const()[name = string("matmul_11_transpose_y_0"), val = bool(false)]; + tensor matmul_11_cast_fp16 = matmul(transpose_x = matmul_11_transpose_x_0, transpose_y = matmul_11_transpose_y_0, x = reshape_33_cast_fp16, y = reshape_34_cast_fp16)[name = string("matmul_11_cast_fp16")]; + tensor concat_215 = const()[name = string("concat_215"), val = tensor([1, 16, 128, 128])]; + tensor reshape_35_cast_fp16 = reshape(shape = concat_215, x = matmul_11_cast_fp16)[name = string("reshape_35_cast_fp16")]; + tensor var_7161_perm_0 = const()[name = string("op_7161_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_7180 = const()[name = string("op_7180"), val = tensor([1, 128, 2048])]; + tensor var_7161_cast_fp16 = transpose(perm = var_7161_perm_0, x = reshape_35_cast_fp16)[name = string("transpose_22")]; + tensor attn_output_115_cast_fp16 = reshape(shape = var_7180, x = var_7161_cast_fp16)[name = string("attn_output_115_cast_fp16")]; + tensor var_7185 = const()[name = string("op_7185"), val = tensor([0, 2, 1])]; + string var_7201_pad_type_0 = const()[name = string("op_7201_pad_type_0"), val = string("valid")]; + int32 var_7201_groups_0 = const()[name = string("op_7201_groups_0"), val = int32(1)]; + tensor var_7201_strides_0 = const()[name = string("op_7201_strides_0"), val = tensor([1])]; + tensor var_7201_pad_0 = const()[name = string("op_7201_pad_0"), val = tensor([0, 0])]; + tensor var_7201_dilations_0 = const()[name = string("op_7201_dilations_0"), val = tensor([1])]; + tensor squeeze_11_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711181312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715375680))))[name = string("squeeze_11_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_7186_cast_fp16 = transpose(perm = var_7185, x = attn_output_115_cast_fp16)[name = string("transpose_21")]; + tensor var_7201_cast_fp16 = conv(dilations = var_7201_dilations_0, groups = var_7201_groups_0, pad = var_7201_pad_0, pad_type = var_7201_pad_type_0, strides = var_7201_strides_0, weight = squeeze_11_cast_fp16_to_fp32_to_fp16_palettized, x = var_7186_cast_fp16)[name = string("op_7201_cast_fp16")]; + tensor var_7205 = const()[name = string("op_7205"), val = tensor([0, 2, 1])]; + tensor attn_output_119_cast_fp16 = transpose(perm = var_7205, x = var_7201_cast_fp16)[name = string("transpose_20")]; + tensor hidden_states_119_cast_fp16 = add(x = hidden_states_111_cast_fp16, y = attn_output_119_cast_fp16)[name = string("hidden_states_119_cast_fp16")]; + int32 var_7218 = const()[name = string("op_7218"), val = int32(-1)]; + fp16 const_405_promoted_to_fp16 = const()[name = string("const_405_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7220_cast_fp16 = mul(x = hidden_states_119_cast_fp16, y = const_405_promoted_to_fp16)[name = string("op_7220_cast_fp16")]; + bool input_209_interleave_0 = const()[name = string("input_209_interleave_0"), val = bool(false)]; + tensor input_209_cast_fp16 = concat(axis = var_7218, interleave = input_209_interleave_0, values = (hidden_states_119_cast_fp16, var_7220_cast_fp16))[name = string("input_209_cast_fp16")]; + tensor normed_189_axes_0 = const()[name = string("normed_189_axes_0"), val = tensor([-1])]; + fp16 var_7215_to_fp16 = const()[name = string("op_7215_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_189_cast_fp16 = layer_norm(axes = normed_189_axes_0, epsilon = var_7215_to_fp16, x = input_209_cast_fp16)[name = string("normed_189_cast_fp16")]; + tensor normed_191_begin_0 = const()[name = string("normed_191_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_191_end_0 = const()[name = string("normed_191_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_191_end_mask_0 = const()[name = string("normed_191_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_191_cast_fp16 = slice_by_index(begin = normed_191_begin_0, end = normed_191_end_0, end_mask = normed_191_end_mask_0, x = normed_189_cast_fp16)[name = string("normed_191_cast_fp16")]; + tensor const_408_promoted_to_fp16 = const()[name = string("const_408_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715506816)))]; + tensor x_189_cast_fp16 = mul(x = normed_191_cast_fp16, y = const_408_promoted_to_fp16)[name = string("x_189_cast_fp16")]; + tensor var_7245 = const()[name = string("op_7245"), val = tensor([0, 2, 1])]; + tensor input_211_axes_0 = const()[name = string("input_211_axes_0"), val = tensor([2])]; + tensor var_7246 = transpose(perm = var_7245, x = x_189_cast_fp16)[name = string("transpose_19")]; + tensor input_211 = expand_dims(axes = input_211_axes_0, x = var_7246)[name = string("input_211")]; + string input_213_pad_type_0 = const()[name = string("input_213_pad_type_0"), val = string("valid")]; + tensor input_213_strides_0 = const()[name = string("input_213_strides_0"), val = tensor([1, 1])]; + tensor input_213_pad_0 = const()[name = string("input_213_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_213_dilations_0 = const()[name = string("input_213_dilations_0"), val = tensor([1, 1])]; + int32 input_213_groups_0 = const()[name = string("input_213_groups_0"), val = int32(1)]; + tensor input_213 = conv(dilations = input_213_dilations_0, groups = input_213_groups_0, pad = input_213_pad_0, pad_type = input_213_pad_type_0, strides = input_213_strides_0, weight = model_model_layers_11_mlp_gate_proj_weight_palettized, x = input_211)[name = string("input_213")]; + string b_23_pad_type_0 = const()[name = string("b_23_pad_type_0"), val = string("valid")]; + tensor b_23_strides_0 = const()[name = string("b_23_strides_0"), val = tensor([1, 1])]; + tensor b_23_pad_0 = const()[name = string("b_23_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_23_dilations_0 = const()[name = string("b_23_dilations_0"), val = tensor([1, 1])]; + int32 b_23_groups_0 = const()[name = string("b_23_groups_0"), val = int32(1)]; + tensor b_23 = conv(dilations = b_23_dilations_0, groups = b_23_groups_0, pad = b_23_pad_0, pad_type = b_23_pad_type_0, strides = b_23_strides_0, weight = model_model_layers_11_mlp_up_proj_weight_palettized, x = input_211)[name = string("b_23")]; + tensor c_23 = silu(x = input_213)[name = string("c_23")]; + tensor input_215 = mul(x = c_23, y = b_23)[name = string("input_215")]; + string e_23_pad_type_0 = const()[name = string("e_23_pad_type_0"), val = string("valid")]; + tensor e_23_strides_0 = const()[name = string("e_23_strides_0"), val = tensor([1, 1])]; + tensor e_23_pad_0 = const()[name = string("e_23_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_23_dilations_0 = const()[name = string("e_23_dilations_0"), val = tensor([1, 1])]; + int32 e_23_groups_0 = const()[name = string("e_23_groups_0"), val = int32(1)]; + tensor e_23 = conv(dilations = e_23_dilations_0, groups = e_23_groups_0, pad = e_23_pad_0, pad_type = e_23_pad_type_0, strides = e_23_strides_0, weight = model_model_layers_11_mlp_down_proj_weight_palettized, x = input_215)[name = string("e_23")]; + tensor var_7268_axes_0 = const()[name = string("op_7268_axes_0"), val = tensor([2])]; + tensor var_7268 = squeeze(axes = var_7268_axes_0, x = e_23)[name = string("op_7268")]; + tensor var_7269 = const()[name = string("op_7269"), val = tensor([0, 2, 1])]; + tensor var_7270 = transpose(perm = var_7269, x = var_7268)[name = string("transpose_18")]; + tensor hidden_states_121_cast_fp16 = add(x = hidden_states_119_cast_fp16, y = var_7270)[name = string("hidden_states_121_cast_fp16")]; + int32 var_7282 = const()[name = string("op_7282"), val = int32(-1)]; + fp16 const_409_promoted_to_fp16 = const()[name = string("const_409_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7284_cast_fp16 = mul(x = hidden_states_121_cast_fp16, y = const_409_promoted_to_fp16)[name = string("op_7284_cast_fp16")]; + bool input_217_interleave_0 = const()[name = string("input_217_interleave_0"), val = bool(false)]; + tensor input_217_cast_fp16 = concat(axis = var_7282, interleave = input_217_interleave_0, values = (hidden_states_121_cast_fp16, var_7284_cast_fp16))[name = string("input_217_cast_fp16")]; + tensor normed_193_axes_0 = const()[name = string("normed_193_axes_0"), val = tensor([-1])]; + fp16 var_7279_to_fp16 = const()[name = string("op_7279_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_193_cast_fp16 = layer_norm(axes = normed_193_axes_0, epsilon = var_7279_to_fp16, x = input_217_cast_fp16)[name = string("normed_193_cast_fp16")]; + tensor normed_195_begin_0 = const()[name = string("normed_195_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_195_end_0 = const()[name = string("normed_195_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_195_end_mask_0 = const()[name = string("normed_195_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_195_cast_fp16 = slice_by_index(begin = normed_195_begin_0, end = normed_195_end_0, end_mask = normed_195_end_mask_0, x = normed_193_cast_fp16)[name = string("normed_195_cast_fp16")]; + tensor const_412_promoted_to_fp16 = const()[name = string("const_412_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715510976)))]; + tensor hidden_states_123_cast_fp16 = mul(x = normed_195_cast_fp16, y = const_412_promoted_to_fp16)[name = string("hidden_states_123_cast_fp16")]; + tensor var_7307 = const()[name = string("op_7307"), val = tensor([0, 2, 1])]; + tensor var_7310_axes_0 = const()[name = string("op_7310_axes_0"), val = tensor([2])]; + tensor var_7308_cast_fp16 = transpose(perm = var_7307, x = hidden_states_123_cast_fp16)[name = string("transpose_17")]; + tensor var_7310_cast_fp16 = expand_dims(axes = var_7310_axes_0, x = var_7308_cast_fp16)[name = string("op_7310_cast_fp16")]; + string query_states_97_pad_type_0 = const()[name = string("query_states_97_pad_type_0"), val = string("valid")]; + tensor query_states_97_strides_0 = const()[name = string("query_states_97_strides_0"), val = tensor([1, 1])]; + tensor query_states_97_pad_0 = const()[name = string("query_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_97_dilations_0 = const()[name = string("query_states_97_dilations_0"), val = tensor([1, 1])]; + int32 query_states_97_groups_0 = const()[name = string("query_states_97_groups_0"), val = int32(1)]; + tensor query_states_97 = conv(dilations = query_states_97_dilations_0, groups = query_states_97_groups_0, pad = query_states_97_pad_0, pad_type = query_states_97_pad_type_0, strides = query_states_97_strides_0, weight = model_model_layers_12_self_attn_q_proj_weight_palettized, x = var_7310_cast_fp16)[name = string("query_states_97")]; + string key_states_121_pad_type_0 = const()[name = string("key_states_121_pad_type_0"), val = string("valid")]; + tensor key_states_121_strides_0 = const()[name = string("key_states_121_strides_0"), val = tensor([1, 1])]; + tensor key_states_121_pad_0 = const()[name = string("key_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_121_dilations_0 = const()[name = string("key_states_121_dilations_0"), val = tensor([1, 1])]; + int32 key_states_121_groups_0 = const()[name = string("key_states_121_groups_0"), val = int32(1)]; + tensor key_states_121 = conv(dilations = key_states_121_dilations_0, groups = key_states_121_groups_0, pad = key_states_121_pad_0, pad_type = key_states_121_pad_type_0, strides = key_states_121_strides_0, weight = model_model_layers_12_self_attn_k_proj_weight_palettized, x = var_7310_cast_fp16)[name = string("key_states_121")]; + string value_states_97_pad_type_0 = const()[name = string("value_states_97_pad_type_0"), val = string("valid")]; + tensor value_states_97_strides_0 = const()[name = string("value_states_97_strides_0"), val = tensor([1, 1])]; + tensor value_states_97_pad_0 = const()[name = string("value_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_97_dilations_0 = const()[name = string("value_states_97_dilations_0"), val = tensor([1, 1])]; + int32 value_states_97_groups_0 = const()[name = string("value_states_97_groups_0"), val = int32(1)]; + tensor value_states_97 = conv(dilations = value_states_97_dilations_0, groups = value_states_97_groups_0, pad = value_states_97_pad_0, pad_type = value_states_97_pad_type_0, strides = value_states_97_strides_0, weight = model_model_layers_12_self_attn_v_proj_weight_palettized, x = var_7310_cast_fp16)[name = string("value_states_97")]; + tensor var_7352 = const()[name = string("op_7352"), val = tensor([1, 16, 128, 128])]; + tensor var_7353 = reshape(shape = var_7352, x = query_states_97)[name = string("op_7353")]; + tensor var_7358 = const()[name = string("op_7358"), val = tensor([0, 1, 3, 2])]; + tensor var_7363 = const()[name = string("op_7363"), val = tensor([1, 8, 128, 128])]; + tensor var_7364 = reshape(shape = var_7363, x = key_states_121)[name = string("op_7364")]; + tensor var_7369 = const()[name = string("op_7369"), val = tensor([0, 1, 3, 2])]; + tensor var_7374 = const()[name = string("op_7374"), val = tensor([1, 8, 128, 128])]; + tensor var_7375 = reshape(shape = var_7374, x = value_states_97)[name = string("op_7375")]; + tensor var_7380 = const()[name = string("op_7380"), val = tensor([0, 1, 3, 2])]; + int32 var_7391 = const()[name = string("op_7391"), val = int32(-1)]; + fp16 const_414_promoted = const()[name = string("const_414_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_125 = transpose(perm = var_7358, x = var_7353)[name = string("transpose_16")]; + tensor var_7393 = mul(x = hidden_states_125, y = const_414_promoted)[name = string("op_7393")]; + bool input_221_interleave_0 = const()[name = string("input_221_interleave_0"), val = bool(false)]; + tensor input_221 = concat(axis = var_7391, interleave = input_221_interleave_0, values = (hidden_states_125, var_7393))[name = string("input_221")]; + tensor normed_197_axes_0 = const()[name = string("normed_197_axes_0"), val = tensor([-1])]; + fp16 var_7388_to_fp16 = const()[name = string("op_7388_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_197_cast_fp16 = layer_norm(axes = normed_197_axes_0, epsilon = var_7388_to_fp16, x = input_221)[name = string("normed_197_cast_fp16")]; + tensor normed_199_begin_0 = const()[name = string("normed_199_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_199_end_0 = const()[name = string("normed_199_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_199_end_mask_0 = const()[name = string("normed_199_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_199 = slice_by_index(begin = normed_199_begin_0, end = normed_199_end_0, end_mask = normed_199_end_mask_0, x = normed_197_cast_fp16)[name = string("normed_199")]; + tensor const_417 = const()[name = string("const_417"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715515136)))]; + tensor q_25 = mul(x = normed_199, y = const_417)[name = string("q_25")]; + int32 var_7416 = const()[name = string("op_7416"), val = int32(-1)]; + fp16 const_418_promoted = const()[name = string("const_418_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_127 = transpose(perm = var_7369, x = var_7364)[name = string("transpose_15")]; + tensor var_7418 = mul(x = hidden_states_127, y = const_418_promoted)[name = string("op_7418")]; + bool input_223_interleave_0 = const()[name = string("input_223_interleave_0"), val = bool(false)]; + tensor input_223 = concat(axis = var_7416, interleave = input_223_interleave_0, values = (hidden_states_127, var_7418))[name = string("input_223")]; + tensor normed_201_axes_0 = const()[name = string("normed_201_axes_0"), val = tensor([-1])]; + fp16 var_7413_to_fp16 = const()[name = string("op_7413_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_201_cast_fp16 = layer_norm(axes = normed_201_axes_0, epsilon = var_7413_to_fp16, x = input_223)[name = string("normed_201_cast_fp16")]; + tensor normed_203_begin_0 = const()[name = string("normed_203_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_203_end_0 = const()[name = string("normed_203_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_203_end_mask_0 = const()[name = string("normed_203_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_203 = slice_by_index(begin = normed_203_begin_0, end = normed_203_end_0, end_mask = normed_203_end_mask_0, x = normed_201_cast_fp16)[name = string("normed_203")]; + tensor const_421 = const()[name = string("const_421"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715515456)))]; + tensor k_25 = mul(x = normed_203, y = const_421)[name = string("k_25")]; + tensor var_7444 = mul(x = q_25, y = cos_5)[name = string("op_7444")]; + tensor x1_49_begin_0 = const()[name = string("x1_49_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_49_end_0 = const()[name = string("x1_49_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_49_end_mask_0 = const()[name = string("x1_49_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_49 = slice_by_index(begin = x1_49_begin_0, end = x1_49_end_0, end_mask = x1_49_end_mask_0, x = q_25)[name = string("x1_49")]; + tensor x2_49_begin_0 = const()[name = string("x2_49_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_49_end_0 = const()[name = string("x2_49_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_49_end_mask_0 = const()[name = string("x2_49_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_49 = slice_by_index(begin = x2_49_begin_0, end = x2_49_end_0, end_mask = x2_49_end_mask_0, x = q_25)[name = string("x2_49")]; + fp16 const_424_promoted = const()[name = string("const_424_promoted"), val = fp16(-0x1p+0)]; + tensor var_7465 = mul(x = x2_49, y = const_424_promoted)[name = string("op_7465")]; + int32 var_7467 = const()[name = string("op_7467"), val = int32(-1)]; + bool var_7468_interleave_0 = const()[name = string("op_7468_interleave_0"), val = bool(false)]; + tensor var_7468 = concat(axis = var_7467, interleave = var_7468_interleave_0, values = (var_7465, x1_49))[name = string("op_7468")]; + tensor var_7469 = mul(x = var_7468, y = sin_5)[name = string("op_7469")]; + tensor query_states_99 = add(x = var_7444, y = var_7469)[name = string("query_states_99")]; + tensor var_7472 = mul(x = k_25, y = cos_5)[name = string("op_7472")]; + tensor x1_51_begin_0 = const()[name = string("x1_51_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_51_end_0 = const()[name = string("x1_51_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_51_end_mask_0 = const()[name = string("x1_51_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_51 = slice_by_index(begin = x1_51_begin_0, end = x1_51_end_0, end_mask = x1_51_end_mask_0, x = k_25)[name = string("x1_51")]; + tensor x2_51_begin_0 = const()[name = string("x2_51_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_51_end_0 = const()[name = string("x2_51_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_51_end_mask_0 = const()[name = string("x2_51_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_51 = slice_by_index(begin = x2_51_begin_0, end = x2_51_end_0, end_mask = x2_51_end_mask_0, x = k_25)[name = string("x2_51")]; + fp16 const_427_promoted = const()[name = string("const_427_promoted"), val = fp16(-0x1p+0)]; + tensor var_7493 = mul(x = x2_51, y = const_427_promoted)[name = string("op_7493")]; + int32 var_7495 = const()[name = string("op_7495"), val = int32(-1)]; + bool var_7496_interleave_0 = const()[name = string("op_7496_interleave_0"), val = bool(false)]; + tensor var_7496 = concat(axis = var_7495, interleave = var_7496_interleave_0, values = (var_7493, x1_51))[name = string("op_7496")]; + tensor var_7497 = mul(x = var_7496, y = sin_5)[name = string("op_7497")]; + tensor key_states_123 = add(x = var_7472, y = var_7497)[name = string("key_states_123")]; + tensor expand_dims_144 = const()[name = string("expand_dims_144"), val = tensor([12])]; + tensor expand_dims_145 = const()[name = string("expand_dims_145"), val = tensor([0])]; + tensor expand_dims_147 = const()[name = string("expand_dims_147"), val = tensor([0])]; + tensor expand_dims_148 = const()[name = string("expand_dims_148"), val = tensor([13])]; + int32 concat_218_axis_0 = const()[name = string("concat_218_axis_0"), val = int32(0)]; + bool concat_218_interleave_0 = const()[name = string("concat_218_interleave_0"), val = bool(false)]; + tensor concat_218 = concat(axis = concat_218_axis_0, interleave = concat_218_interleave_0, values = (expand_dims_144, expand_dims_145, current_pos, expand_dims_147))[name = string("concat_218")]; + tensor concat_219_values1_0 = const()[name = string("concat_219_values1_0"), val = tensor([0])]; + tensor concat_219_values3_0 = const()[name = string("concat_219_values3_0"), val = tensor([0])]; + int32 concat_219_axis_0 = const()[name = string("concat_219_axis_0"), val = int32(0)]; + bool concat_219_interleave_0 = const()[name = string("concat_219_interleave_0"), val = bool(false)]; + tensor concat_219 = concat(axis = concat_219_axis_0, interleave = concat_219_interleave_0, values = (expand_dims_148, concat_219_values1_0, var_1039, concat_219_values3_0))[name = string("concat_219")]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_218, begin_mask = model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0, end = concat_219, end_mask = model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_25_stride_0, update = key_states_123, x = coreml_update_state_51)[name = string("model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_108_write_state")]; + tensor coreml_update_state_52 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_108")]; + tensor expand_dims_150 = const()[name = string("expand_dims_150"), val = tensor([40])]; + tensor expand_dims_151 = const()[name = string("expand_dims_151"), val = tensor([0])]; + tensor expand_dims_153 = const()[name = string("expand_dims_153"), val = tensor([0])]; + tensor expand_dims_154 = const()[name = string("expand_dims_154"), val = tensor([41])]; + int32 concat_222_axis_0 = const()[name = string("concat_222_axis_0"), val = int32(0)]; + bool concat_222_interleave_0 = const()[name = string("concat_222_interleave_0"), val = bool(false)]; + tensor concat_222 = concat(axis = concat_222_axis_0, interleave = concat_222_interleave_0, values = (expand_dims_150, expand_dims_151, current_pos, expand_dims_153))[name = string("concat_222")]; + tensor concat_223_values1_0 = const()[name = string("concat_223_values1_0"), val = tensor([0])]; + tensor concat_223_values3_0 = const()[name = string("concat_223_values3_0"), val = tensor([0])]; + int32 concat_223_axis_0 = const()[name = string("concat_223_axis_0"), val = int32(0)]; + bool concat_223_interleave_0 = const()[name = string("concat_223_interleave_0"), val = bool(false)]; + tensor concat_223 = concat(axis = concat_223_axis_0, interleave = concat_223_interleave_0, values = (expand_dims_154, concat_223_values1_0, var_1039, concat_223_values3_0))[name = string("concat_223")]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_99 = transpose(perm = var_7380, x = var_7375)[name = string("transpose_14")]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_222, begin_mask = model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0, end = concat_223, end_mask = model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_26_stride_0, update = value_states_99, x = coreml_update_state_52)[name = string("model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_109_write_state")]; + tensor coreml_update_state_53 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_109")]; + tensor var_7568_begin_0 = const()[name = string("op_7568_begin_0"), val = tensor([12, 0, 0, 0])]; + tensor var_7568_end_0 = const()[name = string("op_7568_end_0"), val = tensor([13, 8, 1024, 128])]; + tensor var_7568_end_mask_0 = const()[name = string("op_7568_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_7568_cast_fp16 = slice_by_index(begin = var_7568_begin_0, end = var_7568_end_0, end_mask = var_7568_end_mask_0, x = coreml_update_state_53)[name = string("op_7568_cast_fp16")]; + tensor K_layer_cache_25_axes_0 = const()[name = string("K_layer_cache_25_axes_0"), val = tensor([0])]; + tensor K_layer_cache_25_cast_fp16 = squeeze(axes = K_layer_cache_25_axes_0, x = var_7568_cast_fp16)[name = string("K_layer_cache_25_cast_fp16")]; + tensor var_7575_begin_0 = const()[name = string("op_7575_begin_0"), val = tensor([40, 0, 0, 0])]; + tensor var_7575_end_0 = const()[name = string("op_7575_end_0"), val = tensor([41, 8, 1024, 128])]; + tensor var_7575_end_mask_0 = const()[name = string("op_7575_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_7575_cast_fp16 = slice_by_index(begin = var_7575_begin_0, end = var_7575_end_0, end_mask = var_7575_end_mask_0, x = coreml_update_state_53)[name = string("op_7575_cast_fp16")]; + tensor V_layer_cache_25_axes_0 = const()[name = string("V_layer_cache_25_axes_0"), val = tensor([0])]; + tensor V_layer_cache_25_cast_fp16 = squeeze(axes = V_layer_cache_25_axes_0, x = var_7575_cast_fp16)[name = string("V_layer_cache_25_cast_fp16")]; + tensor x_195_axes_0 = const()[name = string("x_195_axes_0"), val = tensor([1])]; + tensor x_195_cast_fp16 = expand_dims(axes = x_195_axes_0, x = K_layer_cache_25_cast_fp16)[name = string("x_195_cast_fp16")]; + tensor var_7604 = const()[name = string("op_7604"), val = tensor([1, 2, 1, 1])]; + tensor x_197_cast_fp16 = tile(reps = var_7604, x = x_195_cast_fp16)[name = string("x_197_cast_fp16")]; + tensor var_7616 = const()[name = string("op_7616"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_127_cast_fp16 = reshape(shape = var_7616, x = x_197_cast_fp16)[name = string("key_states_127_cast_fp16")]; + tensor x_201_axes_0 = const()[name = string("x_201_axes_0"), val = tensor([1])]; + tensor x_201_cast_fp16 = expand_dims(axes = x_201_axes_0, x = V_layer_cache_25_cast_fp16)[name = string("x_201_cast_fp16")]; + tensor var_7624 = const()[name = string("op_7624"), val = tensor([1, 2, 1, 1])]; + tensor x_203_cast_fp16 = tile(reps = var_7624, x = x_201_cast_fp16)[name = string("x_203_cast_fp16")]; + bool var_7651_transpose_x_0 = const()[name = string("op_7651_transpose_x_0"), val = bool(false)]; + bool var_7651_transpose_y_0 = const()[name = string("op_7651_transpose_y_0"), val = bool(true)]; + tensor var_7651 = matmul(transpose_x = var_7651_transpose_x_0, transpose_y = var_7651_transpose_y_0, x = query_states_99, y = key_states_127_cast_fp16)[name = string("op_7651")]; + fp16 var_7652_to_fp16 = const()[name = string("op_7652_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_49_cast_fp16 = mul(x = var_7651, y = var_7652_to_fp16)[name = string("attn_weights_49_cast_fp16")]; + tensor attn_weights_51_cast_fp16 = add(x = attn_weights_49_cast_fp16, y = causal_mask)[name = string("attn_weights_51_cast_fp16")]; + int32 var_7687 = const()[name = string("op_7687"), val = int32(-1)]; + tensor var_7689_cast_fp16 = softmax(axis = var_7687, x = attn_weights_51_cast_fp16)[name = string("op_7689_cast_fp16")]; + tensor concat_228 = const()[name = string("concat_228"), val = tensor([16, 128, 1024])]; + tensor reshape_36_cast_fp16 = reshape(shape = concat_228, x = var_7689_cast_fp16)[name = string("reshape_36_cast_fp16")]; + tensor concat_229 = const()[name = string("concat_229"), val = tensor([16, 1024, 128])]; + tensor reshape_37_cast_fp16 = reshape(shape = concat_229, x = x_203_cast_fp16)[name = string("reshape_37_cast_fp16")]; + bool matmul_12_transpose_x_0 = const()[name = string("matmul_12_transpose_x_0"), val = bool(false)]; + bool matmul_12_transpose_y_0 = const()[name = string("matmul_12_transpose_y_0"), val = bool(false)]; + tensor matmul_12_cast_fp16 = matmul(transpose_x = matmul_12_transpose_x_0, transpose_y = matmul_12_transpose_y_0, x = reshape_36_cast_fp16, y = reshape_37_cast_fp16)[name = string("matmul_12_cast_fp16")]; + tensor concat_233 = const()[name = string("concat_233"), val = tensor([1, 16, 128, 128])]; + tensor reshape_38_cast_fp16 = reshape(shape = concat_233, x = matmul_12_cast_fp16)[name = string("reshape_38_cast_fp16")]; + tensor var_7701_perm_0 = const()[name = string("op_7701_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_7720 = const()[name = string("op_7720"), val = tensor([1, 128, 2048])]; + tensor var_7701_cast_fp16 = transpose(perm = var_7701_perm_0, x = reshape_38_cast_fp16)[name = string("transpose_13")]; + tensor attn_output_125_cast_fp16 = reshape(shape = var_7720, x = var_7701_cast_fp16)[name = string("attn_output_125_cast_fp16")]; + tensor var_7725 = const()[name = string("op_7725"), val = tensor([0, 2, 1])]; + string var_7741_pad_type_0 = const()[name = string("op_7741_pad_type_0"), val = string("valid")]; + int32 var_7741_groups_0 = const()[name = string("op_7741_groups_0"), val = int32(1)]; + tensor var_7741_strides_0 = const()[name = string("op_7741_strides_0"), val = tensor([1])]; + tensor var_7741_pad_0 = const()[name = string("op_7741_pad_0"), val = tensor([0, 0])]; + tensor var_7741_dilations_0 = const()[name = string("op_7741_dilations_0"), val = tensor([1])]; + tensor squeeze_12_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715515776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719710144))))[name = string("squeeze_12_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_7726_cast_fp16 = transpose(perm = var_7725, x = attn_output_125_cast_fp16)[name = string("transpose_12")]; + tensor var_7741_cast_fp16 = conv(dilations = var_7741_dilations_0, groups = var_7741_groups_0, pad = var_7741_pad_0, pad_type = var_7741_pad_type_0, strides = var_7741_strides_0, weight = squeeze_12_cast_fp16_to_fp32_to_fp16_palettized, x = var_7726_cast_fp16)[name = string("op_7741_cast_fp16")]; + tensor var_7745 = const()[name = string("op_7745"), val = tensor([0, 2, 1])]; + tensor attn_output_129_cast_fp16 = transpose(perm = var_7745, x = var_7741_cast_fp16)[name = string("transpose_11")]; + tensor hidden_states_129_cast_fp16 = add(x = hidden_states_121_cast_fp16, y = attn_output_129_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; + int32 var_7758 = const()[name = string("op_7758"), val = int32(-1)]; + fp16 const_439_promoted_to_fp16 = const()[name = string("const_439_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7760_cast_fp16 = mul(x = hidden_states_129_cast_fp16, y = const_439_promoted_to_fp16)[name = string("op_7760_cast_fp16")]; + bool input_227_interleave_0 = const()[name = string("input_227_interleave_0"), val = bool(false)]; + tensor input_227_cast_fp16 = concat(axis = var_7758, interleave = input_227_interleave_0, values = (hidden_states_129_cast_fp16, var_7760_cast_fp16))[name = string("input_227_cast_fp16")]; + tensor normed_205_axes_0 = const()[name = string("normed_205_axes_0"), val = tensor([-1])]; + fp16 var_7755_to_fp16 = const()[name = string("op_7755_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_205_cast_fp16 = layer_norm(axes = normed_205_axes_0, epsilon = var_7755_to_fp16, x = input_227_cast_fp16)[name = string("normed_205_cast_fp16")]; + tensor normed_207_begin_0 = const()[name = string("normed_207_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_207_end_0 = const()[name = string("normed_207_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_207_end_mask_0 = const()[name = string("normed_207_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_207_cast_fp16 = slice_by_index(begin = normed_207_begin_0, end = normed_207_end_0, end_mask = normed_207_end_mask_0, x = normed_205_cast_fp16)[name = string("normed_207_cast_fp16")]; + tensor const_442_promoted_to_fp16 = const()[name = string("const_442_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719841280)))]; + tensor x_205_cast_fp16 = mul(x = normed_207_cast_fp16, y = const_442_promoted_to_fp16)[name = string("x_205_cast_fp16")]; + tensor var_7785 = const()[name = string("op_7785"), val = tensor([0, 2, 1])]; + tensor input_229_axes_0 = const()[name = string("input_229_axes_0"), val = tensor([2])]; + tensor var_7786 = transpose(perm = var_7785, x = x_205_cast_fp16)[name = string("transpose_10")]; + tensor input_229 = expand_dims(axes = input_229_axes_0, x = var_7786)[name = string("input_229")]; + string input_231_pad_type_0 = const()[name = string("input_231_pad_type_0"), val = string("valid")]; + tensor input_231_strides_0 = const()[name = string("input_231_strides_0"), val = tensor([1, 1])]; + tensor input_231_pad_0 = const()[name = string("input_231_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_231_dilations_0 = const()[name = string("input_231_dilations_0"), val = tensor([1, 1])]; + int32 input_231_groups_0 = const()[name = string("input_231_groups_0"), val = int32(1)]; + tensor input_231 = conv(dilations = input_231_dilations_0, groups = input_231_groups_0, pad = input_231_pad_0, pad_type = input_231_pad_type_0, strides = input_231_strides_0, weight = model_model_layers_12_mlp_gate_proj_weight_palettized, x = input_229)[name = string("input_231")]; + string b_25_pad_type_0 = const()[name = string("b_25_pad_type_0"), val = string("valid")]; + tensor b_25_strides_0 = const()[name = string("b_25_strides_0"), val = tensor([1, 1])]; + tensor b_25_pad_0 = const()[name = string("b_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_25_dilations_0 = const()[name = string("b_25_dilations_0"), val = tensor([1, 1])]; + int32 b_25_groups_0 = const()[name = string("b_25_groups_0"), val = int32(1)]; + tensor b_25 = conv(dilations = b_25_dilations_0, groups = b_25_groups_0, pad = b_25_pad_0, pad_type = b_25_pad_type_0, strides = b_25_strides_0, weight = model_model_layers_12_mlp_up_proj_weight_palettized, x = input_229)[name = string("b_25")]; + tensor c_25 = silu(x = input_231)[name = string("c_25")]; + tensor input_233 = mul(x = c_25, y = b_25)[name = string("input_233")]; + string e_25_pad_type_0 = const()[name = string("e_25_pad_type_0"), val = string("valid")]; + tensor e_25_strides_0 = const()[name = string("e_25_strides_0"), val = tensor([1, 1])]; + tensor e_25_pad_0 = const()[name = string("e_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_25_dilations_0 = const()[name = string("e_25_dilations_0"), val = tensor([1, 1])]; + int32 e_25_groups_0 = const()[name = string("e_25_groups_0"), val = int32(1)]; + tensor e_25 = conv(dilations = e_25_dilations_0, groups = e_25_groups_0, pad = e_25_pad_0, pad_type = e_25_pad_type_0, strides = e_25_strides_0, weight = model_model_layers_12_mlp_down_proj_weight_palettized, x = input_233)[name = string("e_25")]; + tensor var_7808_axes_0 = const()[name = string("op_7808_axes_0"), val = tensor([2])]; + tensor var_7808 = squeeze(axes = var_7808_axes_0, x = e_25)[name = string("op_7808")]; + tensor var_7809 = const()[name = string("op_7809"), val = tensor([0, 2, 1])]; + tensor var_7810 = transpose(perm = var_7809, x = var_7808)[name = string("transpose_9")]; + tensor hidden_states_131_cast_fp16 = add(x = hidden_states_129_cast_fp16, y = var_7810)[name = string("hidden_states_131_cast_fp16")]; + int32 var_7822 = const()[name = string("op_7822"), val = int32(-1)]; + fp16 const_443_promoted_to_fp16 = const()[name = string("const_443_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7824_cast_fp16 = mul(x = hidden_states_131_cast_fp16, y = const_443_promoted_to_fp16)[name = string("op_7824_cast_fp16")]; + bool input_235_interleave_0 = const()[name = string("input_235_interleave_0"), val = bool(false)]; + tensor input_235_cast_fp16 = concat(axis = var_7822, interleave = input_235_interleave_0, values = (hidden_states_131_cast_fp16, var_7824_cast_fp16))[name = string("input_235_cast_fp16")]; + tensor normed_209_axes_0 = const()[name = string("normed_209_axes_0"), val = tensor([-1])]; + fp16 var_7819_to_fp16 = const()[name = string("op_7819_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_209_cast_fp16 = layer_norm(axes = normed_209_axes_0, epsilon = var_7819_to_fp16, x = input_235_cast_fp16)[name = string("normed_209_cast_fp16")]; + tensor normed_211_begin_0 = const()[name = string("normed_211_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_211_end_0 = const()[name = string("normed_211_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_211_end_mask_0 = const()[name = string("normed_211_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_211_cast_fp16 = slice_by_index(begin = normed_211_begin_0, end = normed_211_end_0, end_mask = normed_211_end_mask_0, x = normed_209_cast_fp16)[name = string("normed_211_cast_fp16")]; + tensor const_446_promoted_to_fp16 = const()[name = string("const_446_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719845440)))]; + tensor hidden_states_133_cast_fp16 = mul(x = normed_211_cast_fp16, y = const_446_promoted_to_fp16)[name = string("hidden_states_133_cast_fp16")]; + tensor var_7847 = const()[name = string("op_7847"), val = tensor([0, 2, 1])]; + tensor var_7850_axes_0 = const()[name = string("op_7850_axes_0"), val = tensor([2])]; + tensor var_7848_cast_fp16 = transpose(perm = var_7847, x = hidden_states_133_cast_fp16)[name = string("transpose_8")]; + tensor var_7850_cast_fp16 = expand_dims(axes = var_7850_axes_0, x = var_7848_cast_fp16)[name = string("op_7850_cast_fp16")]; + string query_states_105_pad_type_0 = const()[name = string("query_states_105_pad_type_0"), val = string("valid")]; + tensor query_states_105_strides_0 = const()[name = string("query_states_105_strides_0"), val = tensor([1, 1])]; + tensor query_states_105_pad_0 = const()[name = string("query_states_105_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_105_dilations_0 = const()[name = string("query_states_105_dilations_0"), val = tensor([1, 1])]; + int32 query_states_105_groups_0 = const()[name = string("query_states_105_groups_0"), val = int32(1)]; + tensor query_states_105 = conv(dilations = query_states_105_dilations_0, groups = query_states_105_groups_0, pad = query_states_105_pad_0, pad_type = query_states_105_pad_type_0, strides = query_states_105_strides_0, weight = model_model_layers_13_self_attn_q_proj_weight_palettized, x = var_7850_cast_fp16)[name = string("query_states_105")]; + string key_states_131_pad_type_0 = const()[name = string("key_states_131_pad_type_0"), val = string("valid")]; + tensor key_states_131_strides_0 = const()[name = string("key_states_131_strides_0"), val = tensor([1, 1])]; + tensor key_states_131_pad_0 = const()[name = string("key_states_131_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_131_dilations_0 = const()[name = string("key_states_131_dilations_0"), val = tensor([1, 1])]; + int32 key_states_131_groups_0 = const()[name = string("key_states_131_groups_0"), val = int32(1)]; + tensor key_states_131 = conv(dilations = key_states_131_dilations_0, groups = key_states_131_groups_0, pad = key_states_131_pad_0, pad_type = key_states_131_pad_type_0, strides = key_states_131_strides_0, weight = model_model_layers_13_self_attn_k_proj_weight_palettized, x = var_7850_cast_fp16)[name = string("key_states_131")]; + string value_states_105_pad_type_0 = const()[name = string("value_states_105_pad_type_0"), val = string("valid")]; + tensor value_states_105_strides_0 = const()[name = string("value_states_105_strides_0"), val = tensor([1, 1])]; + tensor value_states_105_pad_0 = const()[name = string("value_states_105_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_105_dilations_0 = const()[name = string("value_states_105_dilations_0"), val = tensor([1, 1])]; + int32 value_states_105_groups_0 = const()[name = string("value_states_105_groups_0"), val = int32(1)]; + tensor value_states_105 = conv(dilations = value_states_105_dilations_0, groups = value_states_105_groups_0, pad = value_states_105_pad_0, pad_type = value_states_105_pad_type_0, strides = value_states_105_strides_0, weight = model_model_layers_13_self_attn_v_proj_weight_palettized, x = var_7850_cast_fp16)[name = string("value_states_105")]; + tensor var_7892 = const()[name = string("op_7892"), val = tensor([1, 16, 128, 128])]; + tensor var_7893 = reshape(shape = var_7892, x = query_states_105)[name = string("op_7893")]; + tensor var_7898 = const()[name = string("op_7898"), val = tensor([0, 1, 3, 2])]; + tensor var_7903 = const()[name = string("op_7903"), val = tensor([1, 8, 128, 128])]; + tensor var_7904 = reshape(shape = var_7903, x = key_states_131)[name = string("op_7904")]; + tensor var_7909 = const()[name = string("op_7909"), val = tensor([0, 1, 3, 2])]; + tensor var_7914 = const()[name = string("op_7914"), val = tensor([1, 8, 128, 128])]; + tensor var_7915 = reshape(shape = var_7914, x = value_states_105)[name = string("op_7915")]; + tensor var_7920 = const()[name = string("op_7920"), val = tensor([0, 1, 3, 2])]; + int32 var_7931 = const()[name = string("op_7931"), val = int32(-1)]; + fp16 const_448_promoted = const()[name = string("const_448_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_135 = transpose(perm = var_7898, x = var_7893)[name = string("transpose_7")]; + tensor var_7933 = mul(x = hidden_states_135, y = const_448_promoted)[name = string("op_7933")]; + bool input_239_interleave_0 = const()[name = string("input_239_interleave_0"), val = bool(false)]; + tensor input_239 = concat(axis = var_7931, interleave = input_239_interleave_0, values = (hidden_states_135, var_7933))[name = string("input_239")]; + tensor normed_213_axes_0 = const()[name = string("normed_213_axes_0"), val = tensor([-1])]; + fp16 var_7928_to_fp16 = const()[name = string("op_7928_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_213_cast_fp16 = layer_norm(axes = normed_213_axes_0, epsilon = var_7928_to_fp16, x = input_239)[name = string("normed_213_cast_fp16")]; + tensor normed_215_begin_0 = const()[name = string("normed_215_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_215_end_0 = const()[name = string("normed_215_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_215_end_mask_0 = const()[name = string("normed_215_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_215 = slice_by_index(begin = normed_215_begin_0, end = normed_215_end_0, end_mask = normed_215_end_mask_0, x = normed_213_cast_fp16)[name = string("normed_215")]; + tensor const_451 = const()[name = string("const_451"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719849600)))]; + tensor q = mul(x = normed_215, y = const_451)[name = string("q")]; + int32 var_7956 = const()[name = string("op_7956"), val = int32(-1)]; + fp16 const_452_promoted = const()[name = string("const_452_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_137 = transpose(perm = var_7909, x = var_7904)[name = string("transpose_6")]; + tensor var_7958 = mul(x = hidden_states_137, y = const_452_promoted)[name = string("op_7958")]; + bool input_241_interleave_0 = const()[name = string("input_241_interleave_0"), val = bool(false)]; + tensor input_241 = concat(axis = var_7956, interleave = input_241_interleave_0, values = (hidden_states_137, var_7958))[name = string("input_241")]; + tensor normed_217_axes_0 = const()[name = string("normed_217_axes_0"), val = tensor([-1])]; + fp16 var_7953_to_fp16 = const()[name = string("op_7953_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_217_cast_fp16 = layer_norm(axes = normed_217_axes_0, epsilon = var_7953_to_fp16, x = input_241)[name = string("normed_217_cast_fp16")]; + tensor normed_219_begin_0 = const()[name = string("normed_219_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_219_end_0 = const()[name = string("normed_219_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_219_end_mask_0 = const()[name = string("normed_219_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_219 = slice_by_index(begin = normed_219_begin_0, end = normed_219_end_0, end_mask = normed_219_end_mask_0, x = normed_217_cast_fp16)[name = string("normed_219")]; + tensor const_455 = const()[name = string("const_455"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719849920)))]; + tensor k = mul(x = normed_219, y = const_455)[name = string("k")]; + tensor var_7984 = mul(x = q, y = cos_5)[name = string("op_7984")]; + tensor x1_53_begin_0 = const()[name = string("x1_53_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_53_end_0 = const()[name = string("x1_53_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_53_end_mask_0 = const()[name = string("x1_53_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_53 = slice_by_index(begin = x1_53_begin_0, end = x1_53_end_0, end_mask = x1_53_end_mask_0, x = q)[name = string("x1_53")]; + tensor x2_53_begin_0 = const()[name = string("x2_53_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_53_end_0 = const()[name = string("x2_53_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_53_end_mask_0 = const()[name = string("x2_53_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_53 = slice_by_index(begin = x2_53_begin_0, end = x2_53_end_0, end_mask = x2_53_end_mask_0, x = q)[name = string("x2_53")]; + fp16 const_458_promoted = const()[name = string("const_458_promoted"), val = fp16(-0x1p+0)]; + tensor var_8005 = mul(x = x2_53, y = const_458_promoted)[name = string("op_8005")]; + int32 var_8007 = const()[name = string("op_8007"), val = int32(-1)]; + bool var_8008_interleave_0 = const()[name = string("op_8008_interleave_0"), val = bool(false)]; + tensor var_8008 = concat(axis = var_8007, interleave = var_8008_interleave_0, values = (var_8005, x1_53))[name = string("op_8008")]; + tensor var_8009 = mul(x = var_8008, y = sin_5)[name = string("op_8009")]; + tensor query_states_107 = add(x = var_7984, y = var_8009)[name = string("query_states_107")]; + tensor var_8012 = mul(x = k, y = cos_5)[name = string("op_8012")]; + tensor x1_begin_0 = const()[name = string("x1_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_end_0 = const()[name = string("x1_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_end_mask_0 = const()[name = string("x1_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1 = slice_by_index(begin = x1_begin_0, end = x1_end_0, end_mask = x1_end_mask_0, x = k)[name = string("x1")]; + tensor x2_begin_0 = const()[name = string("x2_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_end_0 = const()[name = string("x2_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_end_mask_0 = const()[name = string("x2_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2 = slice_by_index(begin = x2_begin_0, end = x2_end_0, end_mask = x2_end_mask_0, x = k)[name = string("x2")]; + fp16 const_461_promoted = const()[name = string("const_461_promoted"), val = fp16(-0x1p+0)]; + tensor var_8033 = mul(x = x2, y = const_461_promoted)[name = string("op_8033")]; + int32 var_8035 = const()[name = string("op_8035"), val = int32(-1)]; + bool var_8036_interleave_0 = const()[name = string("op_8036_interleave_0"), val = bool(false)]; + tensor var_8036 = concat(axis = var_8035, interleave = var_8036_interleave_0, values = (var_8033, x1))[name = string("op_8036")]; + tensor var_8037 = mul(x = var_8036, y = sin_5)[name = string("op_8037")]; + tensor key_states_133 = add(x = var_8012, y = var_8037)[name = string("key_states_133")]; + tensor expand_dims_156 = const()[name = string("expand_dims_156"), val = tensor([13])]; + tensor expand_dims_157 = const()[name = string("expand_dims_157"), val = tensor([0])]; + tensor expand_dims_159 = const()[name = string("expand_dims_159"), val = tensor([0])]; + tensor expand_dims_160 = const()[name = string("expand_dims_160"), val = tensor([14])]; + int32 concat_236_axis_0 = const()[name = string("concat_236_axis_0"), val = int32(0)]; + bool concat_236_interleave_0 = const()[name = string("concat_236_interleave_0"), val = bool(false)]; + tensor concat_236 = concat(axis = concat_236_axis_0, interleave = concat_236_interleave_0, values = (expand_dims_156, expand_dims_157, current_pos, expand_dims_159))[name = string("concat_236")]; + tensor concat_237_values1_0 = const()[name = string("concat_237_values1_0"), val = tensor([0])]; + tensor concat_237_values3_0 = const()[name = string("concat_237_values3_0"), val = tensor([0])]; + int32 concat_237_axis_0 = const()[name = string("concat_237_axis_0"), val = int32(0)]; + bool concat_237_interleave_0 = const()[name = string("concat_237_interleave_0"), val = bool(false)]; + tensor concat_237 = concat(axis = concat_237_axis_0, interleave = concat_237_interleave_0, values = (expand_dims_160, concat_237_values1_0, var_1039, concat_237_values3_0))[name = string("concat_237")]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_236, begin_mask = model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0, end = concat_237, end_mask = model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_27_stride_0, update = key_states_133, x = coreml_update_state_53)[name = string("model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_110_write_state")]; + tensor coreml_update_state_54 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_110")]; + tensor expand_dims_162 = const()[name = string("expand_dims_162"), val = tensor([41])]; + tensor expand_dims_163 = const()[name = string("expand_dims_163"), val = tensor([0])]; + tensor expand_dims_165 = const()[name = string("expand_dims_165"), val = tensor([0])]; + tensor expand_dims_166 = const()[name = string("expand_dims_166"), val = tensor([42])]; + int32 concat_240_axis_0 = const()[name = string("concat_240_axis_0"), val = int32(0)]; + bool concat_240_interleave_0 = const()[name = string("concat_240_interleave_0"), val = bool(false)]; + tensor concat_240 = concat(axis = concat_240_axis_0, interleave = concat_240_interleave_0, values = (expand_dims_162, expand_dims_163, current_pos, expand_dims_165))[name = string("concat_240")]; + tensor concat_241_values1_0 = const()[name = string("concat_241_values1_0"), val = tensor([0])]; + tensor concat_241_values3_0 = const()[name = string("concat_241_values3_0"), val = tensor([0])]; + int32 concat_241_axis_0 = const()[name = string("concat_241_axis_0"), val = int32(0)]; + bool concat_241_interleave_0 = const()[name = string("concat_241_interleave_0"), val = bool(false)]; + tensor concat_241 = concat(axis = concat_241_axis_0, interleave = concat_241_interleave_0, values = (expand_dims_166, concat_241_values1_0, var_1039, concat_241_values3_0))[name = string("concat_241")]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_107 = transpose(perm = var_7920, x = var_7915)[name = string("transpose_5")]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_240, begin_mask = model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0, end = concat_241, end_mask = model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_28_stride_0, update = value_states_107, x = coreml_update_state_54)[name = string("model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_111_write_state")]; + tensor coreml_update_state_55 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_111")]; + tensor var_8108_begin_0 = const()[name = string("op_8108_begin_0"), val = tensor([13, 0, 0, 0])]; + tensor var_8108_end_0 = const()[name = string("op_8108_end_0"), val = tensor([14, 8, 1024, 128])]; + tensor var_8108_end_mask_0 = const()[name = string("op_8108_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_8108_cast_fp16 = slice_by_index(begin = var_8108_begin_0, end = var_8108_end_0, end_mask = var_8108_end_mask_0, x = coreml_update_state_55)[name = string("op_8108_cast_fp16")]; + tensor K_layer_cache_axes_0 = const()[name = string("K_layer_cache_axes_0"), val = tensor([0])]; + tensor K_layer_cache_cast_fp16 = squeeze(axes = K_layer_cache_axes_0, x = var_8108_cast_fp16)[name = string("K_layer_cache_cast_fp16")]; + tensor var_8115_begin_0 = const()[name = string("op_8115_begin_0"), val = tensor([41, 0, 0, 0])]; + tensor var_8115_end_0 = const()[name = string("op_8115_end_0"), val = tensor([42, 8, 1024, 128])]; + tensor var_8115_end_mask_0 = const()[name = string("op_8115_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_8115_cast_fp16 = slice_by_index(begin = var_8115_begin_0, end = var_8115_end_0, end_mask = var_8115_end_mask_0, x = coreml_update_state_55)[name = string("op_8115_cast_fp16")]; + tensor V_layer_cache_axes_0 = const()[name = string("V_layer_cache_axes_0"), val = tensor([0])]; + tensor V_layer_cache_cast_fp16 = squeeze(axes = V_layer_cache_axes_0, x = var_8115_cast_fp16)[name = string("V_layer_cache_cast_fp16")]; + tensor x_211_axes_0 = const()[name = string("x_211_axes_0"), val = tensor([1])]; + tensor x_211_cast_fp16 = expand_dims(axes = x_211_axes_0, x = K_layer_cache_cast_fp16)[name = string("x_211_cast_fp16")]; + tensor var_8144 = const()[name = string("op_8144"), val = tensor([1, 2, 1, 1])]; + tensor x_213_cast_fp16 = tile(reps = var_8144, x = x_211_cast_fp16)[name = string("x_213_cast_fp16")]; + tensor var_8156 = const()[name = string("op_8156"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_137_cast_fp16 = reshape(shape = var_8156, x = x_213_cast_fp16)[name = string("key_states_137_cast_fp16")]; + tensor x_217_axes_0 = const()[name = string("x_217_axes_0"), val = tensor([1])]; + tensor x_217_cast_fp16 = expand_dims(axes = x_217_axes_0, x = V_layer_cache_cast_fp16)[name = string("x_217_cast_fp16")]; + tensor var_8164 = const()[name = string("op_8164"), val = tensor([1, 2, 1, 1])]; + tensor x_219_cast_fp16 = tile(reps = var_8164, x = x_217_cast_fp16)[name = string("x_219_cast_fp16")]; + bool var_8191_transpose_x_0 = const()[name = string("op_8191_transpose_x_0"), val = bool(false)]; + bool var_8191_transpose_y_0 = const()[name = string("op_8191_transpose_y_0"), val = bool(true)]; + tensor var_8191 = matmul(transpose_x = var_8191_transpose_x_0, transpose_y = var_8191_transpose_y_0, x = query_states_107, y = key_states_137_cast_fp16)[name = string("op_8191")]; + fp16 var_8192_to_fp16 = const()[name = string("op_8192_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_53_cast_fp16 = mul(x = var_8191, y = var_8192_to_fp16)[name = string("attn_weights_53_cast_fp16")]; + tensor attn_weights_cast_fp16 = add(x = attn_weights_53_cast_fp16, y = causal_mask)[name = string("attn_weights_cast_fp16")]; + int32 var_8227 = const()[name = string("op_8227"), val = int32(-1)]; + tensor var_8229_cast_fp16 = softmax(axis = var_8227, x = attn_weights_cast_fp16)[name = string("op_8229_cast_fp16")]; + tensor concat_246 = const()[name = string("concat_246"), val = tensor([16, 128, 1024])]; + tensor reshape_39_cast_fp16 = reshape(shape = concat_246, x = var_8229_cast_fp16)[name = string("reshape_39_cast_fp16")]; + tensor concat_247 = const()[name = string("concat_247"), val = tensor([16, 1024, 128])]; + tensor reshape_40_cast_fp16 = reshape(shape = concat_247, x = x_219_cast_fp16)[name = string("reshape_40_cast_fp16")]; + bool matmul_13_transpose_x_0 = const()[name = string("matmul_13_transpose_x_0"), val = bool(false)]; + bool matmul_13_transpose_y_0 = const()[name = string("matmul_13_transpose_y_0"), val = bool(false)]; + tensor matmul_13_cast_fp16 = matmul(transpose_x = matmul_13_transpose_x_0, transpose_y = matmul_13_transpose_y_0, x = reshape_39_cast_fp16, y = reshape_40_cast_fp16)[name = string("matmul_13_cast_fp16")]; + tensor concat_251 = const()[name = string("concat_251"), val = tensor([1, 16, 128, 128])]; + tensor reshape_41_cast_fp16 = reshape(shape = concat_251, x = matmul_13_cast_fp16)[name = string("reshape_41_cast_fp16")]; + tensor var_8241_perm_0 = const()[name = string("op_8241_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_8260 = const()[name = string("op_8260"), val = tensor([1, 128, 2048])]; + tensor var_8241_cast_fp16 = transpose(perm = var_8241_perm_0, x = reshape_41_cast_fp16)[name = string("transpose_4")]; + tensor attn_output_135_cast_fp16 = reshape(shape = var_8260, x = var_8241_cast_fp16)[name = string("attn_output_135_cast_fp16")]; + tensor var_8265 = const()[name = string("op_8265"), val = tensor([0, 2, 1])]; + string var_8281_pad_type_0 = const()[name = string("op_8281_pad_type_0"), val = string("valid")]; + int32 var_8281_groups_0 = const()[name = string("op_8281_groups_0"), val = int32(1)]; + tensor var_8281_strides_0 = const()[name = string("op_8281_strides_0"), val = tensor([1])]; + tensor var_8281_pad_0 = const()[name = string("op_8281_pad_0"), val = tensor([0, 0])]; + tensor var_8281_dilations_0 = const()[name = string("op_8281_dilations_0"), val = tensor([1])]; + tensor squeeze_13_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719850240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724044608))))[name = string("squeeze_13_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_8266_cast_fp16 = transpose(perm = var_8265, x = attn_output_135_cast_fp16)[name = string("transpose_3")]; + tensor var_8281_cast_fp16 = conv(dilations = var_8281_dilations_0, groups = var_8281_groups_0, pad = var_8281_pad_0, pad_type = var_8281_pad_type_0, strides = var_8281_strides_0, weight = squeeze_13_cast_fp16_to_fp32_to_fp16_palettized, x = var_8266_cast_fp16)[name = string("op_8281_cast_fp16")]; + tensor var_8285 = const()[name = string("op_8285"), val = tensor([0, 2, 1])]; + tensor attn_output_cast_fp16 = transpose(perm = var_8285, x = var_8281_cast_fp16)[name = string("transpose_2")]; + tensor hidden_states_cast_fp16 = add(x = hidden_states_131_cast_fp16, y = attn_output_cast_fp16)[name = string("hidden_states_cast_fp16")]; + int32 var_8298 = const()[name = string("op_8298"), val = int32(-1)]; + fp16 const_473_promoted_to_fp16 = const()[name = string("const_473_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8300_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_473_promoted_to_fp16)[name = string("op_8300_cast_fp16")]; + bool input_245_interleave_0 = const()[name = string("input_245_interleave_0"), val = bool(false)]; + tensor input_245_cast_fp16 = concat(axis = var_8298, interleave = input_245_interleave_0, values = (hidden_states_cast_fp16, var_8300_cast_fp16))[name = string("input_245_cast_fp16")]; + tensor normed_221_axes_0 = const()[name = string("normed_221_axes_0"), val = tensor([-1])]; + fp16 var_8295_to_fp16 = const()[name = string("op_8295_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_221_cast_fp16 = layer_norm(axes = normed_221_axes_0, epsilon = var_8295_to_fp16, x = input_245_cast_fp16)[name = string("normed_221_cast_fp16")]; + tensor normed_begin_0 = const()[name = string("normed_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_end_0 = const()[name = string("normed_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_end_mask_0 = const()[name = string("normed_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_cast_fp16 = slice_by_index(begin = normed_begin_0, end = normed_end_0, end_mask = normed_end_mask_0, x = normed_221_cast_fp16)[name = string("normed_cast_fp16")]; + tensor const_476_promoted_to_fp16 = const()[name = string("const_476_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724175744)))]; + tensor x_221_cast_fp16 = mul(x = normed_cast_fp16, y = const_476_promoted_to_fp16)[name = string("x_221_cast_fp16")]; + tensor var_8325 = const()[name = string("op_8325"), val = tensor([0, 2, 1])]; + tensor input_247_axes_0 = const()[name = string("input_247_axes_0"), val = tensor([2])]; + tensor var_8326 = transpose(perm = var_8325, x = x_221_cast_fp16)[name = string("transpose_1")]; + tensor input_247 = expand_dims(axes = input_247_axes_0, x = var_8326)[name = string("input_247")]; + string input_249_pad_type_0 = const()[name = string("input_249_pad_type_0"), val = string("valid")]; + tensor input_249_strides_0 = const()[name = string("input_249_strides_0"), val = tensor([1, 1])]; + tensor input_249_pad_0 = const()[name = string("input_249_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_249_dilations_0 = const()[name = string("input_249_dilations_0"), val = tensor([1, 1])]; + int32 input_249_groups_0 = const()[name = string("input_249_groups_0"), val = int32(1)]; + tensor input_249 = conv(dilations = input_249_dilations_0, groups = input_249_groups_0, pad = input_249_pad_0, pad_type = input_249_pad_type_0, strides = input_249_strides_0, weight = model_model_layers_13_mlp_gate_proj_weight_palettized, x = input_247)[name = string("input_249")]; + string b_pad_type_0 = const()[name = string("b_pad_type_0"), val = string("valid")]; + tensor b_strides_0 = const()[name = string("b_strides_0"), val = tensor([1, 1])]; + tensor b_pad_0 = const()[name = string("b_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_dilations_0 = const()[name = string("b_dilations_0"), val = tensor([1, 1])]; + int32 b_groups_0 = const()[name = string("b_groups_0"), val = int32(1)]; + tensor b = conv(dilations = b_dilations_0, groups = b_groups_0, pad = b_pad_0, pad_type = b_pad_type_0, strides = b_strides_0, weight = model_model_layers_13_mlp_up_proj_weight_palettized, x = input_247)[name = string("b")]; + tensor c = silu(x = input_249)[name = string("c")]; + tensor input = mul(x = c, y = b)[name = string("input")]; + string e_pad_type_0 = const()[name = string("e_pad_type_0"), val = string("valid")]; + tensor e_strides_0 = const()[name = string("e_strides_0"), val = tensor([1, 1])]; + tensor e_pad_0 = const()[name = string("e_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_dilations_0 = const()[name = string("e_dilations_0"), val = tensor([1, 1])]; + int32 e_groups_0 = const()[name = string("e_groups_0"), val = int32(1)]; + tensor e = conv(dilations = e_dilations_0, groups = e_groups_0, pad = e_pad_0, pad_type = e_pad_type_0, strides = e_strides_0, weight = model_model_layers_13_mlp_down_proj_weight_palettized, x = input)[name = string("e")]; + tensor var_8348_axes_0 = const()[name = string("op_8348_axes_0"), val = tensor([2])]; + tensor var_8348 = squeeze(axes = var_8348_axes_0, x = e)[name = string("op_8348")]; + tensor var_8349 = const()[name = string("op_8349"), val = tensor([0, 2, 1])]; + tensor var_8350 = transpose(perm = var_8349, x = var_8348)[name = string("transpose_0")]; + tensor output_hidden_states = add(x = hidden_states_cast_fp16, y = var_8350)[name = string("op_8352_cast_fp16")]; + } -> (output_hidden_states); +} \ No newline at end of file