diff --git "a/r2t2_FFN_PF_lut8_chunk_02of02.mlmodelc/model.mil" "b/r2t2_FFN_PF_lut8_chunk_02of02.mlmodelc/model.mil" new file mode 100644--- /dev/null +++ "b/r2t2_FFN_PF_lut8_chunk_02of02.mlmodelc/model.mil" @@ -0,0 +1,6975 @@ +program(1.3) +[buildInfo = dict({{"coremlc-component-MIL", "3600.16.1"}, {"coremlc-version", "3600.25.1"}})] +{ + func infer(tensor causal_mask, tensor current_pos, tensor hidden_states, state> model_model_kv_cache_0, tensor position_ids) { + tensor model_model_layers_14_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4194432))))[name = string("model_model_layers_14_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_14_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4325568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6422784))))[name = string("model_model_layers_14_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_14_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6488384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8585600))))[name = string("model_model_layers_14_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_14_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8651200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21234176))))[name = string("model_model_layers_14_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_14_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21627456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34210432))))[name = string("model_model_layers_14_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_14_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34603712))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47186688))))[name = string("model_model_layers_14_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_15_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47317824))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51512192))))[name = string("model_model_layers_15_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_15_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51643328))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53740544))))[name = string("model_model_layers_15_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_15_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53806144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55903360))))[name = string("model_model_layers_15_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_15_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55968960))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68551936))))[name = string("model_model_layers_15_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_15_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68945216))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(81528192))))[name = string("model_model_layers_15_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_15_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(81921472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(94504448))))[name = string("model_model_layers_15_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_16_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(94635584))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(98829952))))[name = string("model_model_layers_16_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_16_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(98961088))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101058304))))[name = string("model_model_layers_16_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_16_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101123904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103221120))))[name = string("model_model_layers_16_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_16_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103286720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115869696))))[name = string("model_model_layers_16_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_16_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116262976))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(128845952))))[name = string("model_model_layers_16_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_16_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(129239232))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141822208))))[name = string("model_model_layers_16_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_17_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141953344))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146147712))))[name = string("model_model_layers_17_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_17_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146278848))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148376064))))[name = string("model_model_layers_17_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_17_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148441664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(150538880))))[name = string("model_model_layers_17_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_17_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(150604480))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163187456))))[name = string("model_model_layers_17_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_17_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163580736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176163712))))[name = string("model_model_layers_17_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_17_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176556992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(189139968))))[name = string("model_model_layers_17_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_18_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(189271104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193465472))))[name = string("model_model_layers_18_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_18_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193596608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(195693824))))[name = string("model_model_layers_18_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_18_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(195759424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(197856640))))[name = string("model_model_layers_18_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_18_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(197922240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210505216))))[name = string("model_model_layers_18_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_18_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210898496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223481472))))[name = string("model_model_layers_18_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_18_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223874752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236457728))))[name = string("model_model_layers_18_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_19_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236588864))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240783232))))[name = string("model_model_layers_19_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_19_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240914368))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(243011584))))[name = string("model_model_layers_19_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_19_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(243077184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245174400))))[name = string("model_model_layers_19_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_19_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245240000))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(257822976))))[name = string("model_model_layers_19_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_19_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258216256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(270799232))))[name = string("model_model_layers_19_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_19_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(271192512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283775488))))[name = string("model_model_layers_19_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_20_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283906624))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288100992))))[name = string("model_model_layers_20_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_20_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288232128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290329344))))[name = string("model_model_layers_20_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_20_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290394944))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292492160))))[name = string("model_model_layers_20_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_20_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292557760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305140736))))[name = string("model_model_layers_20_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_20_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305534016))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318116992))))[name = string("model_model_layers_20_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_20_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318510272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331093248))))[name = string("model_model_layers_20_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_21_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331224384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335418752))))[name = string("model_model_layers_21_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_21_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335549888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(337647104))))[name = string("model_model_layers_21_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_21_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(337712704))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339809920))))[name = string("model_model_layers_21_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_21_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339875520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352458496))))[name = string("model_model_layers_21_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_21_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352851776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(365434752))))[name = string("model_model_layers_21_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_21_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(365828032))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(378411008))))[name = string("model_model_layers_21_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_22_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(378542144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(382736512))))[name = string("model_model_layers_22_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_22_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(382867648))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(384964864))))[name = string("model_model_layers_22_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_22_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385030464))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(387127680))))[name = string("model_model_layers_22_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_22_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(387193280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399776256))))[name = string("model_model_layers_22_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_22_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400169536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(412752512))))[name = string("model_model_layers_22_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_22_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(413145792))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(425728768))))[name = string("model_model_layers_22_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_23_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(425859904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(430054272))))[name = string("model_model_layers_23_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_23_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(430185408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432282624))))[name = string("model_model_layers_23_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_23_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432348224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(434445440))))[name = string("model_model_layers_23_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_23_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(434511040))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(447094016))))[name = string("model_model_layers_23_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_23_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(447487296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(460070272))))[name = string("model_model_layers_23_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_23_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(460463552))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(473046528))))[name = string("model_model_layers_23_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_24_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(473177664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(477372032))))[name = string("model_model_layers_24_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_24_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(477503168))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479600384))))[name = string("model_model_layers_24_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_24_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479665984))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(481763200))))[name = string("model_model_layers_24_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_24_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(481828800))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(494411776))))[name = string("model_model_layers_24_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_24_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(494805056))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(507388032))))[name = string("model_model_layers_24_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_24_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(507781312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(520364288))))[name = string("model_model_layers_24_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_25_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(520495424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524689792))))[name = string("model_model_layers_25_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_25_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524820928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526918144))))[name = string("model_model_layers_25_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_25_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526983744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(529080960))))[name = string("model_model_layers_25_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_25_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(529146560))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(541729536))))[name = string("model_model_layers_25_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_25_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(542122816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554705792))))[name = string("model_model_layers_25_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_25_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(555099072))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(567682048))))[name = string("model_model_layers_25_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_26_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(567813184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(572007552))))[name = string("model_model_layers_26_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_26_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(572138688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(574235904))))[name = string("model_model_layers_26_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_26_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(574301504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(576398720))))[name = string("model_model_layers_26_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_26_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(576464320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(589047296))))[name = string("model_model_layers_26_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_26_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(589440576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(602023552))))[name = string("model_model_layers_26_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_26_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(602416832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(614999808))))[name = string("model_model_layers_26_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_27_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(615130944))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(619325312))))[name = string("model_model_layers_27_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_27_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(619456448))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(621553664))))[name = string("model_model_layers_27_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_27_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(621619264))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(623716480))))[name = string("model_model_layers_27_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_27_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(623782080))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(636365056))))[name = string("model_model_layers_27_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_27_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(636758336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(649341312))))[name = string("model_model_layers_27_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_27_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(649734592))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(662317568))))[name = string("model_model_layers_27_mlp_down_proj_weight_palettized")]; + int32 var_761_batch_dims_0 = const()[name = string("op_761_batch_dims_0"), val = int32(0)]; + bool var_761_validate_indices_0 = const()[name = string("op_761_validate_indices_0"), val = bool(false)]; + tensor var_753_to_fp16 = const()[name = string("op_753_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(662448704)))]; + string current_pos_to_int16_dtype_0 = const()[name = string("current_pos_to_int16_dtype_0"), val = string("int16")]; + string cast_118_dtype_0 = const()[name = string("cast_118_dtype_0"), val = string("int32")]; + int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; + tensor current_pos_to_int16 = cast(dtype = current_pos_to_int16_dtype_0, x = current_pos)[name = string("cast_5")]; + tensor cast_118 = cast(dtype = cast_118_dtype_0, x = current_pos_to_int16)[name = string("cast_4")]; + tensor greater_equal_0 = greater_equal(x = cast_118, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; + int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(2048)]; + tensor add_0 = add(x = cast_118, y = slice_by_index_0)[name = string("add_0")]; + tensor select_0 = select(a = cast_118, b = add_0, cond = greater_equal_0)[name = string("select_0")]; + string select_0_to_int16_dtype_0 = const()[name = string("select_0_to_int16_dtype_0"), val = string("int16")]; + string cast_0_dtype_0 = const()[name = string("cast_0_dtype_0"), val = string("int32")]; + int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; + tensor select_0_to_int16 = cast(dtype = select_0_to_int16_dtype_0, x = select_0)[name = string("cast_3")]; + tensor cast_0 = cast(dtype = cast_0_dtype_0, x = select_0_to_int16)[name = string("cast_2")]; + tensor greater_equal_0_1 = greater_equal(x = cast_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; + int32 slice_by_index_0_1 = const()[name = string("slice_by_index_0_1"), val = int32(2048)]; + tensor add_0_1 = add(x = cast_0, y = slice_by_index_0_1)[name = string("add_0_1")]; + tensor select_0_1 = select(a = cast_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; + int32 op_761_cast_fp16_cast_uint16_cast_uint16_axis_0 = const()[name = string("op_761_cast_fp16_cast_uint16_cast_uint16_axis_0"), val = int32(1)]; + tensor op_761_cast_fp16_cast_uint16_cast_uint16 = gather(axis = op_761_cast_fp16_cast_uint16_cast_uint16_axis_0, batch_dims = var_761_batch_dims_0, indices = select_0_1, validate_indices = var_761_validate_indices_0, x = var_753_to_fp16)[name = string("op_761_cast_fp16_cast_uint16_cast_uint16")]; + tensor var_766 = const()[name = string("op_766"), val = tensor([1, 1, 1, -1])]; + tensor sin_1_cast_fp16 = reshape(shape = var_766, x = op_761_cast_fp16_cast_uint16_cast_uint16)[name = string("sin_1_cast_fp16")]; + int32 var_776_axis_0 = const()[name = string("op_776_axis_0"), val = int32(1)]; + int32 var_776_batch_dims_0 = const()[name = string("op_776_batch_dims_0"), val = int32(0)]; + bool var_776_validate_indices_0 = const()[name = string("op_776_validate_indices_0"), val = bool(false)]; + tensor var_768_to_fp16 = const()[name = string("op_768_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(662973056)))]; + string current_pos_to_uint16_dtype_0 = const()[name = string("current_pos_to_uint16_dtype_0"), val = string("uint16")]; + tensor current_pos_to_uint16 = cast(dtype = current_pos_to_uint16_dtype_0, x = current_pos)[name = string("cast_1")]; + tensor var_776_cast_fp16_cast_uint16 = gather(axis = var_776_axis_0, batch_dims = var_776_batch_dims_0, indices = current_pos_to_uint16, validate_indices = var_776_validate_indices_0, x = var_768_to_fp16)[name = string("op_776_cast_fp16_cast_uint16")]; + tensor var_781 = const()[name = string("op_781"), val = tensor([1, 1, 1, -1])]; + tensor cos_1_cast_fp16 = reshape(shape = var_781, x = var_776_cast_fp16_cast_uint16)[name = string("cos_1_cast_fp16")]; + int32 var_802 = const()[name = string("op_802"), val = int32(-1)]; + fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_804_cast_fp16 = mul(x = hidden_states, y = const_0_promoted_to_fp16)[name = string("op_804_cast_fp16")]; + bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; + tensor input_1_cast_fp16 = concat(axis = var_802, interleave = input_1_interleave_0, values = (hidden_states, var_804_cast_fp16))[name = string("input_1_cast_fp16")]; + tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; + fp16 var_799_to_fp16 = const()[name = string("op_799_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_799_to_fp16, x = input_1_cast_fp16)[name = string("normed_1_cast_fp16")]; + tensor normed_3_begin_0 = const()[name = string("normed_3_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_3_end_0 = const()[name = string("normed_3_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_3_end_mask_0 = const()[name = string("normed_3_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_3_cast_fp16 = slice_by_index(begin = normed_3_begin_0, end = normed_3_end_0, end_mask = normed_3_end_mask_0, x = normed_1_cast_fp16)[name = string("normed_3_cast_fp16")]; + tensor const_3_promoted_to_fp16 = const()[name = string("const_3_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(663497408)))]; + tensor hidden_states_3_cast_fp16 = mul(x = normed_3_cast_fp16, y = const_3_promoted_to_fp16)[name = string("hidden_states_3_cast_fp16")]; + tensor var_821 = const()[name = string("op_821"), val = tensor([0, 2, 1])]; + tensor var_824_axes_0 = const()[name = string("op_824_axes_0"), val = tensor([2])]; + tensor var_822_cast_fp16 = transpose(perm = var_821, x = hidden_states_3_cast_fp16)[name = string("transpose_83")]; + tensor var_824_cast_fp16 = expand_dims(axes = var_824_axes_0, x = var_822_cast_fp16)[name = string("op_824_cast_fp16")]; + string var_840_pad_type_0 = const()[name = string("op_840_pad_type_0"), val = string("valid")]; + tensor var_840_strides_0 = const()[name = string("op_840_strides_0"), val = tensor([1, 1])]; + tensor var_840_pad_0 = const()[name = string("op_840_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_840_dilations_0 = const()[name = string("op_840_dilations_0"), val = tensor([1, 1])]; + int32 var_840_groups_0 = const()[name = string("op_840_groups_0"), val = int32(1)]; + tensor var_840 = conv(dilations = var_840_dilations_0, groups = var_840_groups_0, pad = var_840_pad_0, pad_type = var_840_pad_type_0, strides = var_840_strides_0, weight = model_model_layers_14_self_attn_q_proj_weight_palettized, x = var_824_cast_fp16)[name = string("op_840")]; + tensor var_845 = const()[name = string("op_845"), val = tensor([1, 16, 1, 128])]; + tensor var_846 = reshape(shape = var_845, x = var_840)[name = string("op_846")]; + string var_862_pad_type_0 = const()[name = string("op_862_pad_type_0"), val = string("valid")]; + tensor var_862_strides_0 = const()[name = string("op_862_strides_0"), val = tensor([1, 1])]; + tensor var_862_pad_0 = const()[name = string("op_862_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_862_dilations_0 = const()[name = string("op_862_dilations_0"), val = tensor([1, 1])]; + int32 var_862_groups_0 = const()[name = string("op_862_groups_0"), val = int32(1)]; + tensor var_862 = conv(dilations = var_862_dilations_0, groups = var_862_groups_0, pad = var_862_pad_0, pad_type = var_862_pad_type_0, strides = var_862_strides_0, weight = model_model_layers_14_self_attn_k_proj_weight_palettized, x = var_824_cast_fp16)[name = string("op_862")]; + tensor var_867 = const()[name = string("op_867"), val = tensor([1, 8, 1, 128])]; + tensor var_868 = reshape(shape = var_867, x = var_862)[name = string("op_868")]; + string var_884_pad_type_0 = const()[name = string("op_884_pad_type_0"), val = string("valid")]; + tensor var_884_strides_0 = const()[name = string("op_884_strides_0"), val = tensor([1, 1])]; + tensor var_884_pad_0 = const()[name = string("op_884_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_884_dilations_0 = const()[name = string("op_884_dilations_0"), val = tensor([1, 1])]; + int32 var_884_groups_0 = const()[name = string("op_884_groups_0"), val = int32(1)]; + tensor var_884 = conv(dilations = var_884_dilations_0, groups = var_884_groups_0, pad = var_884_pad_0, pad_type = var_884_pad_type_0, strides = var_884_strides_0, weight = model_model_layers_14_self_attn_v_proj_weight_palettized, x = var_824_cast_fp16)[name = string("op_884")]; + tensor var_889 = const()[name = string("op_889"), val = tensor([1, 8, 1, 128])]; + tensor var_890 = reshape(shape = var_889, x = var_884)[name = string("op_890")]; + int32 var_905 = const()[name = string("op_905"), val = int32(-1)]; + fp16 const_4_promoted = const()[name = string("const_4_promoted"), val = fp16(-0x1p+0)]; + tensor var_907 = mul(x = var_846, y = const_4_promoted)[name = string("op_907")]; + bool input_5_interleave_0 = const()[name = string("input_5_interleave_0"), val = bool(false)]; + tensor input_5 = concat(axis = var_905, interleave = input_5_interleave_0, values = (var_846, var_907))[name = string("input_5")]; + tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; + fp16 var_902_to_fp16 = const()[name = string("op_902_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_902_to_fp16, x = input_5)[name = string("normed_5_cast_fp16")]; + tensor normed_7_begin_0 = const()[name = string("normed_7_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_7_end_0 = const()[name = string("normed_7_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_7_end_mask_0 = const()[name = string("normed_7_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_7 = slice_by_index(begin = normed_7_begin_0, end = normed_7_end_0, end_mask = normed_7_end_mask_0, x = normed_5_cast_fp16)[name = string("normed_7")]; + tensor const_7 = const()[name = string("const_7"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(663501568)))]; + tensor q_1 = mul(x = normed_7, y = const_7)[name = string("q_1")]; + int32 var_930 = const()[name = string("op_930"), val = int32(-1)]; + fp16 const_8_promoted = const()[name = string("const_8_promoted"), val = fp16(-0x1p+0)]; + tensor var_932 = mul(x = var_868, y = const_8_promoted)[name = string("op_932")]; + bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; + tensor input_7 = concat(axis = var_930, interleave = input_7_interleave_0, values = (var_868, var_932))[name = string("input_7")]; + tensor normed_9_axes_0 = const()[name = string("normed_9_axes_0"), val = tensor([-1])]; + fp16 var_927_to_fp16 = const()[name = string("op_927_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_9_cast_fp16 = layer_norm(axes = normed_9_axes_0, epsilon = var_927_to_fp16, x = input_7)[name = string("normed_9_cast_fp16")]; + tensor normed_11_begin_0 = const()[name = string("normed_11_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_11_end_0 = const()[name = string("normed_11_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_11_end_mask_0 = const()[name = string("normed_11_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_11 = slice_by_index(begin = normed_11_begin_0, end = normed_11_end_0, end_mask = normed_11_end_mask_0, x = normed_9_cast_fp16)[name = string("normed_11")]; + tensor const_11 = const()[name = string("const_11"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(663501888)))]; + tensor k_1 = mul(x = normed_11, y = const_11)[name = string("k_1")]; + tensor var_946 = mul(x = q_1, y = cos_1_cast_fp16)[name = string("op_946")]; + tensor x1_1_begin_0 = const()[name = string("x1_1_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_1_end_0 = const()[name = string("x1_1_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_1_end_mask_0 = const()[name = string("x1_1_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_1 = slice_by_index(begin = x1_1_begin_0, end = x1_1_end_0, end_mask = x1_1_end_mask_0, x = q_1)[name = string("x1_1")]; + tensor x2_1_begin_0 = const()[name = string("x2_1_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_1_end_0 = const()[name = string("x2_1_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_1_end_mask_0 = const()[name = string("x2_1_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_1 = slice_by_index(begin = x2_1_begin_0, end = x2_1_end_0, end_mask = x2_1_end_mask_0, x = q_1)[name = string("x2_1")]; + fp16 const_14_promoted = const()[name = string("const_14_promoted"), val = fp16(-0x1p+0)]; + tensor var_967 = mul(x = x2_1, y = const_14_promoted)[name = string("op_967")]; + int32 var_969 = const()[name = string("op_969"), val = int32(-1)]; + bool var_970_interleave_0 = const()[name = string("op_970_interleave_0"), val = bool(false)]; + tensor var_970 = concat(axis = var_969, interleave = var_970_interleave_0, values = (var_967, x1_1))[name = string("op_970")]; + tensor var_971 = mul(x = var_970, y = sin_1_cast_fp16)[name = string("op_971")]; + tensor query_states_1 = add(x = var_946, y = var_971)[name = string("query_states_1")]; + tensor var_974 = mul(x = k_1, y = cos_1_cast_fp16)[name = string("op_974")]; + tensor x1_3_begin_0 = const()[name = string("x1_3_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_3_end_0 = const()[name = string("x1_3_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_3_end_mask_0 = const()[name = string("x1_3_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_3 = slice_by_index(begin = x1_3_begin_0, end = x1_3_end_0, end_mask = x1_3_end_mask_0, x = k_1)[name = string("x1_3")]; + tensor x2_3_begin_0 = const()[name = string("x2_3_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_3_end_0 = const()[name = string("x2_3_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_3_end_mask_0 = const()[name = string("x2_3_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_3 = slice_by_index(begin = x2_3_begin_0, end = x2_3_end_0, end_mask = x2_3_end_mask_0, x = k_1)[name = string("x2_3")]; + fp16 const_17_promoted = const()[name = string("const_17_promoted"), val = fp16(-0x1p+0)]; + tensor var_995 = mul(x = x2_3, y = const_17_promoted)[name = string("op_995")]; + int32 var_997 = const()[name = string("op_997"), val = int32(-1)]; + bool var_998_interleave_0 = const()[name = string("op_998_interleave_0"), val = bool(false)]; + tensor var_998 = concat(axis = var_997, interleave = var_998_interleave_0, values = (var_995, x1_3))[name = string("op_998")]; + tensor var_999 = mul(x = var_998, y = sin_1_cast_fp16)[name = string("op_999")]; + tensor key_states_1 = add(x = var_974, y = var_999)[name = string("key_states_1")]; + int32 var_1003 = const()[name = string("op_1003"), val = int32(1)]; + tensor var_1004 = add(x = current_pos, y = var_1003)[name = string("op_1004")]; + tensor read_state_0 = read_state(input = model_model_kv_cache_0)[name = string("read_state_0")]; + tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([14])]; + tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; + tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; + tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([15])]; + int32 concat_2_axis_0 = const()[name = string("concat_2_axis_0"), val = int32(0)]; + bool concat_2_interleave_0 = const()[name = string("concat_2_interleave_0"), val = bool(false)]; + tensor concat_2 = concat(axis = concat_2_axis_0, interleave = concat_2_interleave_0, values = (expand_dims_0, expand_dims_1, current_pos, expand_dims_3))[name = string("concat_2")]; + tensor concat_3_values1_0 = const()[name = string("concat_3_values1_0"), val = tensor([0])]; + tensor concat_3_values3_0 = const()[name = string("concat_3_values3_0"), val = tensor([0])]; + int32 concat_3_axis_0 = const()[name = string("concat_3_axis_0"), val = int32(0)]; + bool concat_3_interleave_0 = const()[name = string("concat_3_interleave_0"), val = bool(false)]; + tensor concat_3 = concat(axis = concat_3_axis_0, interleave = concat_3_interleave_0, values = (expand_dims_4, concat_3_values1_0, var_1004, concat_3_values3_0))[name = string("concat_3")]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_2, begin_mask = model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0, end = concat_3, end_mask = model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_1_stride_0, update = key_states_1, x = read_state_0)[name = string("model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_56_write_state")]; + tensor coreml_update_state_28 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_56")]; + tensor expand_dims_6 = const()[name = string("expand_dims_6"), val = tensor([42])]; + tensor expand_dims_7 = const()[name = string("expand_dims_7"), val = tensor([0])]; + tensor expand_dims_9 = const()[name = string("expand_dims_9"), val = tensor([0])]; + tensor expand_dims_10 = const()[name = string("expand_dims_10"), val = tensor([43])]; + int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; + bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; + tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_6, expand_dims_7, current_pos, expand_dims_9))[name = string("concat_6")]; + tensor concat_7_values1_0 = const()[name = string("concat_7_values1_0"), val = tensor([0])]; + tensor concat_7_values3_0 = const()[name = string("concat_7_values3_0"), val = tensor([0])]; + int32 concat_7_axis_0 = const()[name = string("concat_7_axis_0"), val = int32(0)]; + bool concat_7_interleave_0 = const()[name = string("concat_7_interleave_0"), val = bool(false)]; + tensor concat_7 = concat(axis = concat_7_axis_0, interleave = concat_7_interleave_0, values = (expand_dims_10, concat_7_values1_0, var_1004, concat_7_values3_0))[name = string("concat_7")]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_6, begin_mask = model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0, end = concat_7, end_mask = model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_2_stride_0, update = var_890, x = coreml_update_state_28)[name = string("model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_57_write_state")]; + tensor coreml_update_state_29 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_57")]; + tensor var_1054_begin_0 = const()[name = string("op_1054_begin_0"), val = tensor([14, 0, 0, 0])]; + tensor var_1054_end_0 = const()[name = string("op_1054_end_0"), val = tensor([15, 8, 1024, 128])]; + tensor var_1054_end_mask_0 = const()[name = string("op_1054_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1054_cast_fp16 = slice_by_index(begin = var_1054_begin_0, end = var_1054_end_0, end_mask = var_1054_end_mask_0, x = coreml_update_state_29)[name = string("op_1054_cast_fp16")]; + tensor K_layer_cache_1_axes_0 = const()[name = string("K_layer_cache_1_axes_0"), val = tensor([0])]; + tensor K_layer_cache_1_cast_fp16 = squeeze(axes = K_layer_cache_1_axes_0, x = var_1054_cast_fp16)[name = string("K_layer_cache_1_cast_fp16")]; + tensor var_1061_begin_0 = const()[name = string("op_1061_begin_0"), val = tensor([42, 0, 0, 0])]; + tensor var_1061_end_0 = const()[name = string("op_1061_end_0"), val = tensor([43, 8, 1024, 128])]; + tensor var_1061_end_mask_0 = const()[name = string("op_1061_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1061_cast_fp16 = slice_by_index(begin = var_1061_begin_0, end = var_1061_end_0, end_mask = var_1061_end_mask_0, x = coreml_update_state_29)[name = string("op_1061_cast_fp16")]; + tensor V_layer_cache_1_axes_0 = const()[name = string("V_layer_cache_1_axes_0"), val = tensor([0])]; + tensor V_layer_cache_1_cast_fp16 = squeeze(axes = V_layer_cache_1_axes_0, x = var_1061_cast_fp16)[name = string("V_layer_cache_1_cast_fp16")]; + tensor x_3_axes_0 = const()[name = string("x_3_axes_0"), val = tensor([1])]; + tensor x_3_cast_fp16 = expand_dims(axes = x_3_axes_0, x = K_layer_cache_1_cast_fp16)[name = string("x_3_cast_fp16")]; + tensor var_1098 = const()[name = string("op_1098"), val = tensor([1, 2, 1, 1])]; + tensor x_5_cast_fp16 = tile(reps = var_1098, x = x_3_cast_fp16)[name = string("x_5_cast_fp16")]; + tensor var_1110 = const()[name = string("op_1110"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_3_cast_fp16 = reshape(shape = var_1110, x = x_5_cast_fp16)[name = string("key_states_3_cast_fp16")]; + tensor x_9_axes_0 = const()[name = string("x_9_axes_0"), val = tensor([1])]; + tensor x_9_cast_fp16 = expand_dims(axes = x_9_axes_0, x = V_layer_cache_1_cast_fp16)[name = string("x_9_cast_fp16")]; + tensor var_1118 = const()[name = string("op_1118"), val = tensor([1, 2, 1, 1])]; + tensor x_11_cast_fp16 = tile(reps = var_1118, x = x_9_cast_fp16)[name = string("x_11_cast_fp16")]; + tensor var_1130 = const()[name = string("op_1130"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_3_cast_fp16 = reshape(shape = var_1130, x = x_11_cast_fp16)[name = string("value_states_3_cast_fp16")]; + bool var_1145_transpose_x_1 = const()[name = string("op_1145_transpose_x_1"), val = bool(false)]; + bool var_1145_transpose_y_1 = const()[name = string("op_1145_transpose_y_1"), val = bool(true)]; + tensor var_1145 = matmul(transpose_x = var_1145_transpose_x_1, transpose_y = var_1145_transpose_y_1, x = query_states_1, y = key_states_3_cast_fp16)[name = string("op_1145")]; + fp16 var_1146_to_fp16 = const()[name = string("op_1146_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_1_cast_fp16 = mul(x = var_1145, y = var_1146_to_fp16)[name = string("attn_weights_1_cast_fp16")]; + tensor attn_weights_3_cast_fp16 = add(x = attn_weights_1_cast_fp16, y = causal_mask)[name = string("attn_weights_3_cast_fp16")]; + int32 var_1181 = const()[name = string("op_1181"), val = int32(-1)]; + tensor attn_weights_5_cast_fp16 = softmax(axis = var_1181, x = attn_weights_3_cast_fp16)[name = string("attn_weights_5_cast_fp16")]; + bool attn_output_1_transpose_x_0 = const()[name = string("attn_output_1_transpose_x_0"), val = bool(false)]; + bool attn_output_1_transpose_y_0 = const()[name = string("attn_output_1_transpose_y_0"), val = bool(false)]; + tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_0, transpose_y = attn_output_1_transpose_y_0, x = attn_weights_5_cast_fp16, y = value_states_3_cast_fp16)[name = string("attn_output_1_cast_fp16")]; + tensor var_1192_perm_0 = const()[name = string("op_1192_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1196 = const()[name = string("op_1196"), val = tensor([1, 1, 2048])]; + tensor var_1192_cast_fp16 = transpose(perm = var_1192_perm_0, x = attn_output_1_cast_fp16)[name = string("transpose_82")]; + tensor attn_output_5_cast_fp16 = reshape(shape = var_1196, x = var_1192_cast_fp16)[name = string("attn_output_5_cast_fp16")]; + tensor var_1201 = const()[name = string("op_1201"), val = tensor([0, 2, 1])]; + string var_1217_pad_type_0 = const()[name = string("op_1217_pad_type_0"), val = string("valid")]; + int32 var_1217_groups_0 = const()[name = string("op_1217_groups_0"), val = int32(1)]; + tensor var_1217_strides_0 = const()[name = string("op_1217_strides_0"), val = tensor([1])]; + tensor var_1217_pad_0 = const()[name = string("op_1217_pad_0"), val = tensor([0, 0])]; + tensor var_1217_dilations_0 = const()[name = string("op_1217_dilations_0"), val = tensor([1])]; + tensor squeeze_0_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(663502208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667696576))))[name = string("squeeze_0_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_1202_cast_fp16 = transpose(perm = var_1201, x = attn_output_5_cast_fp16)[name = string("transpose_81")]; + tensor var_1217_cast_fp16 = conv(dilations = var_1217_dilations_0, groups = var_1217_groups_0, pad = var_1217_pad_0, pad_type = var_1217_pad_type_0, strides = var_1217_strides_0, weight = squeeze_0_cast_fp16_to_fp32_to_fp16_palettized, x = var_1202_cast_fp16)[name = string("op_1217_cast_fp16")]; + tensor var_1221 = const()[name = string("op_1221"), val = tensor([0, 2, 1])]; + tensor attn_output_9_cast_fp16 = transpose(perm = var_1221, x = var_1217_cast_fp16)[name = string("transpose_80")]; + tensor hidden_states_9_cast_fp16 = add(x = hidden_states, y = attn_output_9_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; + int32 var_1234 = const()[name = string("op_1234"), val = int32(-1)]; + fp16 const_26_promoted_to_fp16 = const()[name = string("const_26_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1236_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_26_promoted_to_fp16)[name = string("op_1236_cast_fp16")]; + bool input_11_interleave_0 = const()[name = string("input_11_interleave_0"), val = bool(false)]; + tensor input_11_cast_fp16 = concat(axis = var_1234, interleave = input_11_interleave_0, values = (hidden_states_9_cast_fp16, var_1236_cast_fp16))[name = string("input_11_cast_fp16")]; + tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; + fp16 var_1231_to_fp16 = const()[name = string("op_1231_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_1231_to_fp16, x = input_11_cast_fp16)[name = string("normed_13_cast_fp16")]; + tensor normed_15_begin_0 = const()[name = string("normed_15_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_15_end_0 = const()[name = string("normed_15_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_15_end_mask_0 = const()[name = string("normed_15_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_15_cast_fp16 = slice_by_index(begin = normed_15_begin_0, end = normed_15_end_0, end_mask = normed_15_end_mask_0, x = normed_13_cast_fp16)[name = string("normed_15_cast_fp16")]; + tensor const_29_promoted_to_fp16 = const()[name = string("const_29_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667827712)))]; + tensor x_13_cast_fp16 = mul(x = normed_15_cast_fp16, y = const_29_promoted_to_fp16)[name = string("x_13_cast_fp16")]; + tensor var_1261 = const()[name = string("op_1261"), val = tensor([0, 2, 1])]; + tensor input_13_axes_0 = const()[name = string("input_13_axes_0"), val = tensor([2])]; + tensor var_1262 = transpose(perm = var_1261, x = x_13_cast_fp16)[name = string("transpose_79")]; + tensor input_13 = expand_dims(axes = input_13_axes_0, x = var_1262)[name = string("input_13")]; + string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; + tensor input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor([1, 1])]; + tensor input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor([1, 1])]; + int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; + tensor input_15 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = model_model_layers_14_mlp_gate_proj_weight_palettized, x = input_13)[name = string("input_15")]; + string b_1_pad_type_0 = const()[name = string("b_1_pad_type_0"), val = string("valid")]; + tensor b_1_strides_0 = const()[name = string("b_1_strides_0"), val = tensor([1, 1])]; + tensor b_1_pad_0 = const()[name = string("b_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_1_dilations_0 = const()[name = string("b_1_dilations_0"), val = tensor([1, 1])]; + int32 b_1_groups_0 = const()[name = string("b_1_groups_0"), val = int32(1)]; + tensor b_1 = conv(dilations = b_1_dilations_0, groups = b_1_groups_0, pad = b_1_pad_0, pad_type = b_1_pad_type_0, strides = b_1_strides_0, weight = model_model_layers_14_mlp_up_proj_weight_palettized, x = input_13)[name = string("b_1")]; + tensor c_1 = silu(x = input_15)[name = string("c_1")]; + tensor input_17 = mul(x = c_1, y = b_1)[name = string("input_17")]; + string e_1_pad_type_0 = const()[name = string("e_1_pad_type_0"), val = string("valid")]; + tensor e_1_strides_0 = const()[name = string("e_1_strides_0"), val = tensor([1, 1])]; + tensor e_1_pad_0 = const()[name = string("e_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_1_dilations_0 = const()[name = string("e_1_dilations_0"), val = tensor([1, 1])]; + int32 e_1_groups_0 = const()[name = string("e_1_groups_0"), val = int32(1)]; + tensor e_1 = conv(dilations = e_1_dilations_0, groups = e_1_groups_0, pad = e_1_pad_0, pad_type = e_1_pad_type_0, strides = e_1_strides_0, weight = model_model_layers_14_mlp_down_proj_weight_palettized, x = input_17)[name = string("e_1")]; + tensor var_1284_axes_0 = const()[name = string("op_1284_axes_0"), val = tensor([2])]; + tensor var_1284 = squeeze(axes = var_1284_axes_0, x = e_1)[name = string("op_1284")]; + tensor var_1285 = const()[name = string("op_1285"), val = tensor([0, 2, 1])]; + tensor var_1286 = transpose(perm = var_1285, x = var_1284)[name = string("transpose_78")]; + tensor hidden_states_11_cast_fp16 = add(x = hidden_states_9_cast_fp16, y = var_1286)[name = string("hidden_states_11_cast_fp16")]; + int32 var_1298 = const()[name = string("op_1298"), val = int32(-1)]; + fp16 const_30_promoted_to_fp16 = const()[name = string("const_30_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1300_cast_fp16 = mul(x = hidden_states_11_cast_fp16, y = const_30_promoted_to_fp16)[name = string("op_1300_cast_fp16")]; + bool input_19_interleave_0 = const()[name = string("input_19_interleave_0"), val = bool(false)]; + tensor input_19_cast_fp16 = concat(axis = var_1298, interleave = input_19_interleave_0, values = (hidden_states_11_cast_fp16, var_1300_cast_fp16))[name = string("input_19_cast_fp16")]; + tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; + fp16 var_1295_to_fp16 = const()[name = string("op_1295_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_1295_to_fp16, x = input_19_cast_fp16)[name = string("normed_17_cast_fp16")]; + tensor normed_19_begin_0 = const()[name = string("normed_19_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_19_end_0 = const()[name = string("normed_19_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_19_end_mask_0 = const()[name = string("normed_19_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_19_cast_fp16 = slice_by_index(begin = normed_19_begin_0, end = normed_19_end_0, end_mask = normed_19_end_mask_0, x = normed_17_cast_fp16)[name = string("normed_19_cast_fp16")]; + tensor const_33_promoted_to_fp16 = const()[name = string("const_33_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667831872)))]; + tensor hidden_states_13_cast_fp16 = mul(x = normed_19_cast_fp16, y = const_33_promoted_to_fp16)[name = string("hidden_states_13_cast_fp16")]; + tensor var_1317 = const()[name = string("op_1317"), val = tensor([0, 2, 1])]; + tensor var_1320_axes_0 = const()[name = string("op_1320_axes_0"), val = tensor([2])]; + tensor var_1318_cast_fp16 = transpose(perm = var_1317, x = hidden_states_13_cast_fp16)[name = string("transpose_77")]; + tensor var_1320_cast_fp16 = expand_dims(axes = var_1320_axes_0, x = var_1318_cast_fp16)[name = string("op_1320_cast_fp16")]; + string var_1336_pad_type_0 = const()[name = string("op_1336_pad_type_0"), val = string("valid")]; + tensor var_1336_strides_0 = const()[name = string("op_1336_strides_0"), val = tensor([1, 1])]; + tensor var_1336_pad_0 = const()[name = string("op_1336_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1336_dilations_0 = const()[name = string("op_1336_dilations_0"), val = tensor([1, 1])]; + int32 var_1336_groups_0 = const()[name = string("op_1336_groups_0"), val = int32(1)]; + tensor var_1336 = conv(dilations = var_1336_dilations_0, groups = var_1336_groups_0, pad = var_1336_pad_0, pad_type = var_1336_pad_type_0, strides = var_1336_strides_0, weight = model_model_layers_15_self_attn_q_proj_weight_palettized, x = var_1320_cast_fp16)[name = string("op_1336")]; + tensor var_1341 = const()[name = string("op_1341"), val = tensor([1, 16, 1, 128])]; + tensor var_1342 = reshape(shape = var_1341, x = var_1336)[name = string("op_1342")]; + string var_1358_pad_type_0 = const()[name = string("op_1358_pad_type_0"), val = string("valid")]; + tensor var_1358_strides_0 = const()[name = string("op_1358_strides_0"), val = tensor([1, 1])]; + tensor var_1358_pad_0 = const()[name = string("op_1358_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1358_dilations_0 = const()[name = string("op_1358_dilations_0"), val = tensor([1, 1])]; + int32 var_1358_groups_0 = const()[name = string("op_1358_groups_0"), val = int32(1)]; + tensor var_1358 = conv(dilations = var_1358_dilations_0, groups = var_1358_groups_0, pad = var_1358_pad_0, pad_type = var_1358_pad_type_0, strides = var_1358_strides_0, weight = model_model_layers_15_self_attn_k_proj_weight_palettized, x = var_1320_cast_fp16)[name = string("op_1358")]; + tensor var_1363 = const()[name = string("op_1363"), val = tensor([1, 8, 1, 128])]; + tensor var_1364 = reshape(shape = var_1363, x = var_1358)[name = string("op_1364")]; + string var_1380_pad_type_0 = const()[name = string("op_1380_pad_type_0"), val = string("valid")]; + tensor var_1380_strides_0 = const()[name = string("op_1380_strides_0"), val = tensor([1, 1])]; + tensor var_1380_pad_0 = const()[name = string("op_1380_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1380_dilations_0 = const()[name = string("op_1380_dilations_0"), val = tensor([1, 1])]; + int32 var_1380_groups_0 = const()[name = string("op_1380_groups_0"), val = int32(1)]; + tensor var_1380 = conv(dilations = var_1380_dilations_0, groups = var_1380_groups_0, pad = var_1380_pad_0, pad_type = var_1380_pad_type_0, strides = var_1380_strides_0, weight = model_model_layers_15_self_attn_v_proj_weight_palettized, x = var_1320_cast_fp16)[name = string("op_1380")]; + tensor var_1385 = const()[name = string("op_1385"), val = tensor([1, 8, 1, 128])]; + tensor var_1386 = reshape(shape = var_1385, x = var_1380)[name = string("op_1386")]; + int32 var_1401 = const()[name = string("op_1401"), val = int32(-1)]; + fp16 const_34_promoted = const()[name = string("const_34_promoted"), val = fp16(-0x1p+0)]; + tensor var_1403 = mul(x = var_1342, y = const_34_promoted)[name = string("op_1403")]; + bool input_23_interleave_0 = const()[name = string("input_23_interleave_0"), val = bool(false)]; + tensor input_23 = concat(axis = var_1401, interleave = input_23_interleave_0, values = (var_1342, var_1403))[name = string("input_23")]; + tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; + fp16 var_1398_to_fp16 = const()[name = string("op_1398_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_1398_to_fp16, x = input_23)[name = string("normed_21_cast_fp16")]; + tensor normed_23_begin_0 = const()[name = string("normed_23_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_23_end_0 = const()[name = string("normed_23_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_23_end_mask_0 = const()[name = string("normed_23_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_23 = slice_by_index(begin = normed_23_begin_0, end = normed_23_end_0, end_mask = normed_23_end_mask_0, x = normed_21_cast_fp16)[name = string("normed_23")]; + tensor const_37 = const()[name = string("const_37"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667836032)))]; + tensor q_3 = mul(x = normed_23, y = const_37)[name = string("q_3")]; + int32 var_1426 = const()[name = string("op_1426"), val = int32(-1)]; + fp16 const_38_promoted = const()[name = string("const_38_promoted"), val = fp16(-0x1p+0)]; + tensor var_1428 = mul(x = var_1364, y = const_38_promoted)[name = string("op_1428")]; + bool input_25_interleave_0 = const()[name = string("input_25_interleave_0"), val = bool(false)]; + tensor input_25 = concat(axis = var_1426, interleave = input_25_interleave_0, values = (var_1364, var_1428))[name = string("input_25")]; + tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; + fp16 var_1423_to_fp16 = const()[name = string("op_1423_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_1423_to_fp16, x = input_25)[name = string("normed_25_cast_fp16")]; + tensor normed_27_begin_0 = const()[name = string("normed_27_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_27_end_0 = const()[name = string("normed_27_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_27_end_mask_0 = const()[name = string("normed_27_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_27 = slice_by_index(begin = normed_27_begin_0, end = normed_27_end_0, end_mask = normed_27_end_mask_0, x = normed_25_cast_fp16)[name = string("normed_27")]; + tensor const_41 = const()[name = string("const_41"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667836352)))]; + tensor k_3 = mul(x = normed_27, y = const_41)[name = string("k_3")]; + tensor var_1442 = mul(x = q_3, y = cos_1_cast_fp16)[name = string("op_1442")]; + tensor x1_5_begin_0 = const()[name = string("x1_5_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_5_end_0 = const()[name = string("x1_5_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_5_end_mask_0 = const()[name = string("x1_5_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_5 = slice_by_index(begin = x1_5_begin_0, end = x1_5_end_0, end_mask = x1_5_end_mask_0, x = q_3)[name = string("x1_5")]; + tensor x2_5_begin_0 = const()[name = string("x2_5_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_5_end_0 = const()[name = string("x2_5_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_5_end_mask_0 = const()[name = string("x2_5_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_5 = slice_by_index(begin = x2_5_begin_0, end = x2_5_end_0, end_mask = x2_5_end_mask_0, x = q_3)[name = string("x2_5")]; + fp16 const_44_promoted = const()[name = string("const_44_promoted"), val = fp16(-0x1p+0)]; + tensor var_1463 = mul(x = x2_5, y = const_44_promoted)[name = string("op_1463")]; + int32 var_1465 = const()[name = string("op_1465"), val = int32(-1)]; + bool var_1466_interleave_0 = const()[name = string("op_1466_interleave_0"), val = bool(false)]; + tensor var_1466 = concat(axis = var_1465, interleave = var_1466_interleave_0, values = (var_1463, x1_5))[name = string("op_1466")]; + tensor var_1467 = mul(x = var_1466, y = sin_1_cast_fp16)[name = string("op_1467")]; + tensor query_states_5 = add(x = var_1442, y = var_1467)[name = string("query_states_5")]; + tensor var_1470 = mul(x = k_3, y = cos_1_cast_fp16)[name = string("op_1470")]; + tensor x1_7_begin_0 = const()[name = string("x1_7_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_7_end_0 = const()[name = string("x1_7_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_7_end_mask_0 = const()[name = string("x1_7_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_7 = slice_by_index(begin = x1_7_begin_0, end = x1_7_end_0, end_mask = x1_7_end_mask_0, x = k_3)[name = string("x1_7")]; + tensor x2_7_begin_0 = const()[name = string("x2_7_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_7_end_0 = const()[name = string("x2_7_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_7_end_mask_0 = const()[name = string("x2_7_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_7 = slice_by_index(begin = x2_7_begin_0, end = x2_7_end_0, end_mask = x2_7_end_mask_0, x = k_3)[name = string("x2_7")]; + fp16 const_47_promoted = const()[name = string("const_47_promoted"), val = fp16(-0x1p+0)]; + tensor var_1491 = mul(x = x2_7, y = const_47_promoted)[name = string("op_1491")]; + int32 var_1493 = const()[name = string("op_1493"), val = int32(-1)]; + bool var_1494_interleave_0 = const()[name = string("op_1494_interleave_0"), val = bool(false)]; + tensor var_1494 = concat(axis = var_1493, interleave = var_1494_interleave_0, values = (var_1491, x1_7))[name = string("op_1494")]; + tensor var_1495 = mul(x = var_1494, y = sin_1_cast_fp16)[name = string("op_1495")]; + tensor key_states_5 = add(x = var_1470, y = var_1495)[name = string("key_states_5")]; + tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([15])]; + tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; + tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; + tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([16])]; + int32 concat_10_axis_0 = const()[name = string("concat_10_axis_0"), val = int32(0)]; + bool concat_10_interleave_0 = const()[name = string("concat_10_interleave_0"), val = bool(false)]; + tensor concat_10 = concat(axis = concat_10_axis_0, interleave = concat_10_interleave_0, values = (expand_dims_12, expand_dims_13, current_pos, expand_dims_15))[name = string("concat_10")]; + tensor concat_11_values1_0 = const()[name = string("concat_11_values1_0"), val = tensor([0])]; + tensor concat_11_values3_0 = const()[name = string("concat_11_values3_0"), val = tensor([0])]; + int32 concat_11_axis_0 = const()[name = string("concat_11_axis_0"), val = int32(0)]; + bool concat_11_interleave_0 = const()[name = string("concat_11_interleave_0"), val = bool(false)]; + tensor concat_11 = concat(axis = concat_11_axis_0, interleave = concat_11_interleave_0, values = (expand_dims_16, concat_11_values1_0, var_1004, concat_11_values3_0))[name = string("concat_11")]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_10, begin_mask = model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0, end = concat_11, end_mask = model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_3_stride_0, update = key_states_5, x = coreml_update_state_29)[name = string("model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_58_write_state")]; + tensor coreml_update_state_30 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_58")]; + tensor expand_dims_18 = const()[name = string("expand_dims_18"), val = tensor([43])]; + tensor expand_dims_19 = const()[name = string("expand_dims_19"), val = tensor([0])]; + tensor expand_dims_21 = const()[name = string("expand_dims_21"), val = tensor([0])]; + tensor expand_dims_22 = const()[name = string("expand_dims_22"), val = tensor([44])]; + int32 concat_14_axis_0 = const()[name = string("concat_14_axis_0"), val = int32(0)]; + bool concat_14_interleave_0 = const()[name = string("concat_14_interleave_0"), val = bool(false)]; + tensor concat_14 = concat(axis = concat_14_axis_0, interleave = concat_14_interleave_0, values = (expand_dims_18, expand_dims_19, current_pos, expand_dims_21))[name = string("concat_14")]; + tensor concat_15_values1_0 = const()[name = string("concat_15_values1_0"), val = tensor([0])]; + tensor concat_15_values3_0 = const()[name = string("concat_15_values3_0"), val = tensor([0])]; + int32 concat_15_axis_0 = const()[name = string("concat_15_axis_0"), val = int32(0)]; + bool concat_15_interleave_0 = const()[name = string("concat_15_interleave_0"), val = bool(false)]; + tensor concat_15 = concat(axis = concat_15_axis_0, interleave = concat_15_interleave_0, values = (expand_dims_22, concat_15_values1_0, var_1004, concat_15_values3_0))[name = string("concat_15")]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_14, begin_mask = model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0, end = concat_15, end_mask = model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_4_stride_0, update = var_1386, x = coreml_update_state_30)[name = string("model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_59_write_state")]; + tensor coreml_update_state_31 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_59")]; + tensor var_1550_begin_0 = const()[name = string("op_1550_begin_0"), val = tensor([15, 0, 0, 0])]; + tensor var_1550_end_0 = const()[name = string("op_1550_end_0"), val = tensor([16, 8, 1024, 128])]; + tensor var_1550_end_mask_0 = const()[name = string("op_1550_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1550_cast_fp16 = slice_by_index(begin = var_1550_begin_0, end = var_1550_end_0, end_mask = var_1550_end_mask_0, x = coreml_update_state_31)[name = string("op_1550_cast_fp16")]; + tensor K_layer_cache_3_axes_0 = const()[name = string("K_layer_cache_3_axes_0"), val = tensor([0])]; + tensor K_layer_cache_3_cast_fp16 = squeeze(axes = K_layer_cache_3_axes_0, x = var_1550_cast_fp16)[name = string("K_layer_cache_3_cast_fp16")]; + tensor var_1557_begin_0 = const()[name = string("op_1557_begin_0"), val = tensor([43, 0, 0, 0])]; + tensor var_1557_end_0 = const()[name = string("op_1557_end_0"), val = tensor([44, 8, 1024, 128])]; + tensor var_1557_end_mask_0 = const()[name = string("op_1557_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1557_cast_fp16 = slice_by_index(begin = var_1557_begin_0, end = var_1557_end_0, end_mask = var_1557_end_mask_0, x = coreml_update_state_31)[name = string("op_1557_cast_fp16")]; + tensor V_layer_cache_3_axes_0 = const()[name = string("V_layer_cache_3_axes_0"), val = tensor([0])]; + tensor V_layer_cache_3_cast_fp16 = squeeze(axes = V_layer_cache_3_axes_0, x = var_1557_cast_fp16)[name = string("V_layer_cache_3_cast_fp16")]; + tensor x_19_axes_0 = const()[name = string("x_19_axes_0"), val = tensor([1])]; + tensor x_19_cast_fp16 = expand_dims(axes = x_19_axes_0, x = K_layer_cache_3_cast_fp16)[name = string("x_19_cast_fp16")]; + tensor var_1594 = const()[name = string("op_1594"), val = tensor([1, 2, 1, 1])]; + tensor x_21_cast_fp16 = tile(reps = var_1594, x = x_19_cast_fp16)[name = string("x_21_cast_fp16")]; + tensor var_1606 = const()[name = string("op_1606"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_7_cast_fp16 = reshape(shape = var_1606, x = x_21_cast_fp16)[name = string("key_states_7_cast_fp16")]; + tensor x_25_axes_0 = const()[name = string("x_25_axes_0"), val = tensor([1])]; + tensor x_25_cast_fp16 = expand_dims(axes = x_25_axes_0, x = V_layer_cache_3_cast_fp16)[name = string("x_25_cast_fp16")]; + tensor var_1614 = const()[name = string("op_1614"), val = tensor([1, 2, 1, 1])]; + tensor x_27_cast_fp16 = tile(reps = var_1614, x = x_25_cast_fp16)[name = string("x_27_cast_fp16")]; + tensor var_1626 = const()[name = string("op_1626"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_9_cast_fp16 = reshape(shape = var_1626, x = x_27_cast_fp16)[name = string("value_states_9_cast_fp16")]; + bool var_1641_transpose_x_1 = const()[name = string("op_1641_transpose_x_1"), val = bool(false)]; + bool var_1641_transpose_y_1 = const()[name = string("op_1641_transpose_y_1"), val = bool(true)]; + tensor var_1641 = matmul(transpose_x = var_1641_transpose_x_1, transpose_y = var_1641_transpose_y_1, x = query_states_5, y = key_states_7_cast_fp16)[name = string("op_1641")]; + fp16 var_1642_to_fp16 = const()[name = string("op_1642_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_7_cast_fp16 = mul(x = var_1641, y = var_1642_to_fp16)[name = string("attn_weights_7_cast_fp16")]; + tensor attn_weights_9_cast_fp16 = add(x = attn_weights_7_cast_fp16, y = causal_mask)[name = string("attn_weights_9_cast_fp16")]; + int32 var_1677 = const()[name = string("op_1677"), val = int32(-1)]; + tensor attn_weights_11_cast_fp16 = softmax(axis = var_1677, x = attn_weights_9_cast_fp16)[name = string("attn_weights_11_cast_fp16")]; + bool attn_output_11_transpose_x_0 = const()[name = string("attn_output_11_transpose_x_0"), val = bool(false)]; + bool attn_output_11_transpose_y_0 = const()[name = string("attn_output_11_transpose_y_0"), val = bool(false)]; + tensor attn_output_11_cast_fp16 = matmul(transpose_x = attn_output_11_transpose_x_0, transpose_y = attn_output_11_transpose_y_0, x = attn_weights_11_cast_fp16, y = value_states_9_cast_fp16)[name = string("attn_output_11_cast_fp16")]; + tensor var_1688_perm_0 = const()[name = string("op_1688_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1692 = const()[name = string("op_1692"), val = tensor([1, 1, 2048])]; + tensor var_1688_cast_fp16 = transpose(perm = var_1688_perm_0, x = attn_output_11_cast_fp16)[name = string("transpose_76")]; + tensor attn_output_15_cast_fp16 = reshape(shape = var_1692, x = var_1688_cast_fp16)[name = string("attn_output_15_cast_fp16")]; + tensor var_1697 = const()[name = string("op_1697"), val = tensor([0, 2, 1])]; + string var_1713_pad_type_0 = const()[name = string("op_1713_pad_type_0"), val = string("valid")]; + int32 var_1713_groups_0 = const()[name = string("op_1713_groups_0"), val = int32(1)]; + tensor var_1713_strides_0 = const()[name = string("op_1713_strides_0"), val = tensor([1])]; + tensor var_1713_pad_0 = const()[name = string("op_1713_pad_0"), val = tensor([0, 0])]; + tensor var_1713_dilations_0 = const()[name = string("op_1713_dilations_0"), val = tensor([1])]; + tensor squeeze_1_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667836672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672031040))))[name = string("squeeze_1_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_1698_cast_fp16 = transpose(perm = var_1697, x = attn_output_15_cast_fp16)[name = string("transpose_75")]; + tensor var_1713_cast_fp16 = conv(dilations = var_1713_dilations_0, groups = var_1713_groups_0, pad = var_1713_pad_0, pad_type = var_1713_pad_type_0, strides = var_1713_strides_0, weight = squeeze_1_cast_fp16_to_fp32_to_fp16_palettized, x = var_1698_cast_fp16)[name = string("op_1713_cast_fp16")]; + tensor var_1717 = const()[name = string("op_1717"), val = tensor([0, 2, 1])]; + tensor attn_output_19_cast_fp16 = transpose(perm = var_1717, x = var_1713_cast_fp16)[name = string("transpose_74")]; + tensor hidden_states_19_cast_fp16 = add(x = hidden_states_11_cast_fp16, y = attn_output_19_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; + int32 var_1730 = const()[name = string("op_1730"), val = int32(-1)]; + fp16 const_56_promoted_to_fp16 = const()[name = string("const_56_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1732_cast_fp16 = mul(x = hidden_states_19_cast_fp16, y = const_56_promoted_to_fp16)[name = string("op_1732_cast_fp16")]; + bool input_29_interleave_0 = const()[name = string("input_29_interleave_0"), val = bool(false)]; + tensor input_29_cast_fp16 = concat(axis = var_1730, interleave = input_29_interleave_0, values = (hidden_states_19_cast_fp16, var_1732_cast_fp16))[name = string("input_29_cast_fp16")]; + tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; + fp16 var_1727_to_fp16 = const()[name = string("op_1727_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_1727_to_fp16, x = input_29_cast_fp16)[name = string("normed_29_cast_fp16")]; + tensor normed_31_begin_0 = const()[name = string("normed_31_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_31_end_0 = const()[name = string("normed_31_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_31_end_mask_0 = const()[name = string("normed_31_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_31_cast_fp16 = slice_by_index(begin = normed_31_begin_0, end = normed_31_end_0, end_mask = normed_31_end_mask_0, x = normed_29_cast_fp16)[name = string("normed_31_cast_fp16")]; + tensor const_59_promoted_to_fp16 = const()[name = string("const_59_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672162176)))]; + tensor x_29_cast_fp16 = mul(x = normed_31_cast_fp16, y = const_59_promoted_to_fp16)[name = string("x_29_cast_fp16")]; + tensor var_1757 = const()[name = string("op_1757"), val = tensor([0, 2, 1])]; + tensor input_31_axes_0 = const()[name = string("input_31_axes_0"), val = tensor([2])]; + tensor var_1758 = transpose(perm = var_1757, x = x_29_cast_fp16)[name = string("transpose_73")]; + tensor input_31 = expand_dims(axes = input_31_axes_0, x = var_1758)[name = string("input_31")]; + string input_33_pad_type_0 = const()[name = string("input_33_pad_type_0"), val = string("valid")]; + tensor input_33_strides_0 = const()[name = string("input_33_strides_0"), val = tensor([1, 1])]; + tensor input_33_pad_0 = const()[name = string("input_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_33_dilations_0 = const()[name = string("input_33_dilations_0"), val = tensor([1, 1])]; + int32 input_33_groups_0 = const()[name = string("input_33_groups_0"), val = int32(1)]; + tensor input_33 = conv(dilations = input_33_dilations_0, groups = input_33_groups_0, pad = input_33_pad_0, pad_type = input_33_pad_type_0, strides = input_33_strides_0, weight = model_model_layers_15_mlp_gate_proj_weight_palettized, x = input_31)[name = string("input_33")]; + string b_3_pad_type_0 = const()[name = string("b_3_pad_type_0"), val = string("valid")]; + tensor b_3_strides_0 = const()[name = string("b_3_strides_0"), val = tensor([1, 1])]; + tensor b_3_pad_0 = const()[name = string("b_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_3_dilations_0 = const()[name = string("b_3_dilations_0"), val = tensor([1, 1])]; + int32 b_3_groups_0 = const()[name = string("b_3_groups_0"), val = int32(1)]; + tensor b_3 = conv(dilations = b_3_dilations_0, groups = b_3_groups_0, pad = b_3_pad_0, pad_type = b_3_pad_type_0, strides = b_3_strides_0, weight = model_model_layers_15_mlp_up_proj_weight_palettized, x = input_31)[name = string("b_3")]; + tensor c_3 = silu(x = input_33)[name = string("c_3")]; + tensor input_35 = mul(x = c_3, y = b_3)[name = string("input_35")]; + string e_3_pad_type_0 = const()[name = string("e_3_pad_type_0"), val = string("valid")]; + tensor e_3_strides_0 = const()[name = string("e_3_strides_0"), val = tensor([1, 1])]; + tensor e_3_pad_0 = const()[name = string("e_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_3_dilations_0 = const()[name = string("e_3_dilations_0"), val = tensor([1, 1])]; + int32 e_3_groups_0 = const()[name = string("e_3_groups_0"), val = int32(1)]; + tensor e_3 = conv(dilations = e_3_dilations_0, groups = e_3_groups_0, pad = e_3_pad_0, pad_type = e_3_pad_type_0, strides = e_3_strides_0, weight = model_model_layers_15_mlp_down_proj_weight_palettized, x = input_35)[name = string("e_3")]; + tensor var_1780_axes_0 = const()[name = string("op_1780_axes_0"), val = tensor([2])]; + tensor var_1780 = squeeze(axes = var_1780_axes_0, x = e_3)[name = string("op_1780")]; + tensor var_1781 = const()[name = string("op_1781"), val = tensor([0, 2, 1])]; + tensor var_1782 = transpose(perm = var_1781, x = var_1780)[name = string("transpose_72")]; + tensor hidden_states_21_cast_fp16 = add(x = hidden_states_19_cast_fp16, y = var_1782)[name = string("hidden_states_21_cast_fp16")]; + int32 var_1794 = const()[name = string("op_1794"), val = int32(-1)]; + fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1796_cast_fp16 = mul(x = hidden_states_21_cast_fp16, y = const_60_promoted_to_fp16)[name = string("op_1796_cast_fp16")]; + bool input_37_interleave_0 = const()[name = string("input_37_interleave_0"), val = bool(false)]; + tensor input_37_cast_fp16 = concat(axis = var_1794, interleave = input_37_interleave_0, values = (hidden_states_21_cast_fp16, var_1796_cast_fp16))[name = string("input_37_cast_fp16")]; + tensor normed_33_axes_0 = const()[name = string("normed_33_axes_0"), val = tensor([-1])]; + fp16 var_1791_to_fp16 = const()[name = string("op_1791_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_33_cast_fp16 = layer_norm(axes = normed_33_axes_0, epsilon = var_1791_to_fp16, x = input_37_cast_fp16)[name = string("normed_33_cast_fp16")]; + tensor normed_35_begin_0 = const()[name = string("normed_35_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_35_end_0 = const()[name = string("normed_35_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_35_end_mask_0 = const()[name = string("normed_35_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_35_cast_fp16 = slice_by_index(begin = normed_35_begin_0, end = normed_35_end_0, end_mask = normed_35_end_mask_0, x = normed_33_cast_fp16)[name = string("normed_35_cast_fp16")]; + tensor const_63_promoted_to_fp16 = const()[name = string("const_63_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672166336)))]; + tensor hidden_states_23_cast_fp16 = mul(x = normed_35_cast_fp16, y = const_63_promoted_to_fp16)[name = string("hidden_states_23_cast_fp16")]; + tensor var_1813 = const()[name = string("op_1813"), val = tensor([0, 2, 1])]; + tensor var_1816_axes_0 = const()[name = string("op_1816_axes_0"), val = tensor([2])]; + tensor var_1814_cast_fp16 = transpose(perm = var_1813, x = hidden_states_23_cast_fp16)[name = string("transpose_71")]; + tensor var_1816_cast_fp16 = expand_dims(axes = var_1816_axes_0, x = var_1814_cast_fp16)[name = string("op_1816_cast_fp16")]; + string var_1832_pad_type_0 = const()[name = string("op_1832_pad_type_0"), val = string("valid")]; + tensor var_1832_strides_0 = const()[name = string("op_1832_strides_0"), val = tensor([1, 1])]; + tensor var_1832_pad_0 = const()[name = string("op_1832_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1832_dilations_0 = const()[name = string("op_1832_dilations_0"), val = tensor([1, 1])]; + int32 var_1832_groups_0 = const()[name = string("op_1832_groups_0"), val = int32(1)]; + tensor var_1832 = conv(dilations = var_1832_dilations_0, groups = var_1832_groups_0, pad = var_1832_pad_0, pad_type = var_1832_pad_type_0, strides = var_1832_strides_0, weight = model_model_layers_16_self_attn_q_proj_weight_palettized, x = var_1816_cast_fp16)[name = string("op_1832")]; + tensor var_1837 = const()[name = string("op_1837"), val = tensor([1, 16, 1, 128])]; + tensor var_1838 = reshape(shape = var_1837, x = var_1832)[name = string("op_1838")]; + string var_1854_pad_type_0 = const()[name = string("op_1854_pad_type_0"), val = string("valid")]; + tensor var_1854_strides_0 = const()[name = string("op_1854_strides_0"), val = tensor([1, 1])]; + tensor var_1854_pad_0 = const()[name = string("op_1854_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1854_dilations_0 = const()[name = string("op_1854_dilations_0"), val = tensor([1, 1])]; + int32 var_1854_groups_0 = const()[name = string("op_1854_groups_0"), val = int32(1)]; + tensor var_1854 = conv(dilations = var_1854_dilations_0, groups = var_1854_groups_0, pad = var_1854_pad_0, pad_type = var_1854_pad_type_0, strides = var_1854_strides_0, weight = model_model_layers_16_self_attn_k_proj_weight_palettized, x = var_1816_cast_fp16)[name = string("op_1854")]; + tensor var_1859 = const()[name = string("op_1859"), val = tensor([1, 8, 1, 128])]; + tensor var_1860 = reshape(shape = var_1859, x = var_1854)[name = string("op_1860")]; + string var_1876_pad_type_0 = const()[name = string("op_1876_pad_type_0"), val = string("valid")]; + tensor var_1876_strides_0 = const()[name = string("op_1876_strides_0"), val = tensor([1, 1])]; + tensor var_1876_pad_0 = const()[name = string("op_1876_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_1876_dilations_0 = const()[name = string("op_1876_dilations_0"), val = tensor([1, 1])]; + int32 var_1876_groups_0 = const()[name = string("op_1876_groups_0"), val = int32(1)]; + tensor var_1876 = conv(dilations = var_1876_dilations_0, groups = var_1876_groups_0, pad = var_1876_pad_0, pad_type = var_1876_pad_type_0, strides = var_1876_strides_0, weight = model_model_layers_16_self_attn_v_proj_weight_palettized, x = var_1816_cast_fp16)[name = string("op_1876")]; + tensor var_1881 = const()[name = string("op_1881"), val = tensor([1, 8, 1, 128])]; + tensor var_1882 = reshape(shape = var_1881, x = var_1876)[name = string("op_1882")]; + int32 var_1897 = const()[name = string("op_1897"), val = int32(-1)]; + fp16 const_64_promoted = const()[name = string("const_64_promoted"), val = fp16(-0x1p+0)]; + tensor var_1899 = mul(x = var_1838, y = const_64_promoted)[name = string("op_1899")]; + bool input_41_interleave_0 = const()[name = string("input_41_interleave_0"), val = bool(false)]; + tensor input_41 = concat(axis = var_1897, interleave = input_41_interleave_0, values = (var_1838, var_1899))[name = string("input_41")]; + tensor normed_37_axes_0 = const()[name = string("normed_37_axes_0"), val = tensor([-1])]; + fp16 var_1894_to_fp16 = const()[name = string("op_1894_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_37_cast_fp16 = layer_norm(axes = normed_37_axes_0, epsilon = var_1894_to_fp16, x = input_41)[name = string("normed_37_cast_fp16")]; + tensor normed_39_begin_0 = const()[name = string("normed_39_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_39_end_0 = const()[name = string("normed_39_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_39_end_mask_0 = const()[name = string("normed_39_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_39 = slice_by_index(begin = normed_39_begin_0, end = normed_39_end_0, end_mask = normed_39_end_mask_0, x = normed_37_cast_fp16)[name = string("normed_39")]; + tensor const_67 = const()[name = string("const_67"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672170496)))]; + tensor q_5 = mul(x = normed_39, y = const_67)[name = string("q_5")]; + int32 var_1922 = const()[name = string("op_1922"), val = int32(-1)]; + fp16 const_68_promoted = const()[name = string("const_68_promoted"), val = fp16(-0x1p+0)]; + tensor var_1924 = mul(x = var_1860, y = const_68_promoted)[name = string("op_1924")]; + bool input_43_interleave_0 = const()[name = string("input_43_interleave_0"), val = bool(false)]; + tensor input_43 = concat(axis = var_1922, interleave = input_43_interleave_0, values = (var_1860, var_1924))[name = string("input_43")]; + tensor normed_41_axes_0 = const()[name = string("normed_41_axes_0"), val = tensor([-1])]; + fp16 var_1919_to_fp16 = const()[name = string("op_1919_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_41_cast_fp16 = layer_norm(axes = normed_41_axes_0, epsilon = var_1919_to_fp16, x = input_43)[name = string("normed_41_cast_fp16")]; + tensor normed_43_begin_0 = const()[name = string("normed_43_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_43_end_0 = const()[name = string("normed_43_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_43_end_mask_0 = const()[name = string("normed_43_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_43 = slice_by_index(begin = normed_43_begin_0, end = normed_43_end_0, end_mask = normed_43_end_mask_0, x = normed_41_cast_fp16)[name = string("normed_43")]; + tensor const_71 = const()[name = string("const_71"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672170816)))]; + tensor k_5 = mul(x = normed_43, y = const_71)[name = string("k_5")]; + tensor var_1938 = mul(x = q_5, y = cos_1_cast_fp16)[name = string("op_1938")]; + tensor x1_9_begin_0 = const()[name = string("x1_9_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_9_end_0 = const()[name = string("x1_9_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_9_end_mask_0 = const()[name = string("x1_9_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_9 = slice_by_index(begin = x1_9_begin_0, end = x1_9_end_0, end_mask = x1_9_end_mask_0, x = q_5)[name = string("x1_9")]; + tensor x2_9_begin_0 = const()[name = string("x2_9_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_9_end_0 = const()[name = string("x2_9_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_9_end_mask_0 = const()[name = string("x2_9_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_9 = slice_by_index(begin = x2_9_begin_0, end = x2_9_end_0, end_mask = x2_9_end_mask_0, x = q_5)[name = string("x2_9")]; + fp16 const_74_promoted = const()[name = string("const_74_promoted"), val = fp16(-0x1p+0)]; + tensor var_1959 = mul(x = x2_9, y = const_74_promoted)[name = string("op_1959")]; + int32 var_1961 = const()[name = string("op_1961"), val = int32(-1)]; + bool var_1962_interleave_0 = const()[name = string("op_1962_interleave_0"), val = bool(false)]; + tensor var_1962 = concat(axis = var_1961, interleave = var_1962_interleave_0, values = (var_1959, x1_9))[name = string("op_1962")]; + tensor var_1963 = mul(x = var_1962, y = sin_1_cast_fp16)[name = string("op_1963")]; + tensor query_states_9 = add(x = var_1938, y = var_1963)[name = string("query_states_9")]; + tensor var_1966 = mul(x = k_5, y = cos_1_cast_fp16)[name = string("op_1966")]; + tensor x1_11_begin_0 = const()[name = string("x1_11_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_11_end_0 = const()[name = string("x1_11_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_11_end_mask_0 = const()[name = string("x1_11_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_11 = slice_by_index(begin = x1_11_begin_0, end = x1_11_end_0, end_mask = x1_11_end_mask_0, x = k_5)[name = string("x1_11")]; + tensor x2_11_begin_0 = const()[name = string("x2_11_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_11_end_0 = const()[name = string("x2_11_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_11_end_mask_0 = const()[name = string("x2_11_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_11 = slice_by_index(begin = x2_11_begin_0, end = x2_11_end_0, end_mask = x2_11_end_mask_0, x = k_5)[name = string("x2_11")]; + fp16 const_77_promoted = const()[name = string("const_77_promoted"), val = fp16(-0x1p+0)]; + tensor var_1987 = mul(x = x2_11, y = const_77_promoted)[name = string("op_1987")]; + int32 var_1989 = const()[name = string("op_1989"), val = int32(-1)]; + bool var_1990_interleave_0 = const()[name = string("op_1990_interleave_0"), val = bool(false)]; + tensor var_1990 = concat(axis = var_1989, interleave = var_1990_interleave_0, values = (var_1987, x1_11))[name = string("op_1990")]; + tensor var_1991 = mul(x = var_1990, y = sin_1_cast_fp16)[name = string("op_1991")]; + tensor key_states_9 = add(x = var_1966, y = var_1991)[name = string("key_states_9")]; + tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([16])]; + tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; + tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; + tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([17])]; + int32 concat_18_axis_0 = const()[name = string("concat_18_axis_0"), val = int32(0)]; + bool concat_18_interleave_0 = const()[name = string("concat_18_interleave_0"), val = bool(false)]; + tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (expand_dims_24, expand_dims_25, current_pos, expand_dims_27))[name = string("concat_18")]; + tensor concat_19_values1_0 = const()[name = string("concat_19_values1_0"), val = tensor([0])]; + tensor concat_19_values3_0 = const()[name = string("concat_19_values3_0"), val = tensor([0])]; + int32 concat_19_axis_0 = const()[name = string("concat_19_axis_0"), val = int32(0)]; + bool concat_19_interleave_0 = const()[name = string("concat_19_interleave_0"), val = bool(false)]; + tensor concat_19 = concat(axis = concat_19_axis_0, interleave = concat_19_interleave_0, values = (expand_dims_28, concat_19_values1_0, var_1004, concat_19_values3_0))[name = string("concat_19")]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_18, begin_mask = model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0, end = concat_19, end_mask = model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_5_stride_0, update = key_states_9, x = coreml_update_state_31)[name = string("model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_60_write_state")]; + tensor coreml_update_state_32 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_60")]; + tensor expand_dims_30 = const()[name = string("expand_dims_30"), val = tensor([44])]; + tensor expand_dims_31 = const()[name = string("expand_dims_31"), val = tensor([0])]; + tensor expand_dims_33 = const()[name = string("expand_dims_33"), val = tensor([0])]; + tensor expand_dims_34 = const()[name = string("expand_dims_34"), val = tensor([45])]; + int32 concat_22_axis_0 = const()[name = string("concat_22_axis_0"), val = int32(0)]; + bool concat_22_interleave_0 = const()[name = string("concat_22_interleave_0"), val = bool(false)]; + tensor concat_22 = concat(axis = concat_22_axis_0, interleave = concat_22_interleave_0, values = (expand_dims_30, expand_dims_31, current_pos, expand_dims_33))[name = string("concat_22")]; + tensor concat_23_values1_0 = const()[name = string("concat_23_values1_0"), val = tensor([0])]; + tensor concat_23_values3_0 = const()[name = string("concat_23_values3_0"), val = tensor([0])]; + int32 concat_23_axis_0 = const()[name = string("concat_23_axis_0"), val = int32(0)]; + bool concat_23_interleave_0 = const()[name = string("concat_23_interleave_0"), val = bool(false)]; + tensor concat_23 = concat(axis = concat_23_axis_0, interleave = concat_23_interleave_0, values = (expand_dims_34, concat_23_values1_0, var_1004, concat_23_values3_0))[name = string("concat_23")]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_22, begin_mask = model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0, end = concat_23, end_mask = model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_6_stride_0, update = var_1882, x = coreml_update_state_32)[name = string("model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_61_write_state")]; + tensor coreml_update_state_33 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_61")]; + tensor var_2046_begin_0 = const()[name = string("op_2046_begin_0"), val = tensor([16, 0, 0, 0])]; + tensor var_2046_end_0 = const()[name = string("op_2046_end_0"), val = tensor([17, 8, 1024, 128])]; + tensor var_2046_end_mask_0 = const()[name = string("op_2046_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2046_cast_fp16 = slice_by_index(begin = var_2046_begin_0, end = var_2046_end_0, end_mask = var_2046_end_mask_0, x = coreml_update_state_33)[name = string("op_2046_cast_fp16")]; + tensor K_layer_cache_5_axes_0 = const()[name = string("K_layer_cache_5_axes_0"), val = tensor([0])]; + tensor K_layer_cache_5_cast_fp16 = squeeze(axes = K_layer_cache_5_axes_0, x = var_2046_cast_fp16)[name = string("K_layer_cache_5_cast_fp16")]; + tensor var_2053_begin_0 = const()[name = string("op_2053_begin_0"), val = tensor([44, 0, 0, 0])]; + tensor var_2053_end_0 = const()[name = string("op_2053_end_0"), val = tensor([45, 8, 1024, 128])]; + tensor var_2053_end_mask_0 = const()[name = string("op_2053_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2053_cast_fp16 = slice_by_index(begin = var_2053_begin_0, end = var_2053_end_0, end_mask = var_2053_end_mask_0, x = coreml_update_state_33)[name = string("op_2053_cast_fp16")]; + tensor V_layer_cache_5_axes_0 = const()[name = string("V_layer_cache_5_axes_0"), val = tensor([0])]; + tensor V_layer_cache_5_cast_fp16 = squeeze(axes = V_layer_cache_5_axes_0, x = var_2053_cast_fp16)[name = string("V_layer_cache_5_cast_fp16")]; + tensor x_35_axes_0 = const()[name = string("x_35_axes_0"), val = tensor([1])]; + tensor x_35_cast_fp16 = expand_dims(axes = x_35_axes_0, x = K_layer_cache_5_cast_fp16)[name = string("x_35_cast_fp16")]; + tensor var_2090 = const()[name = string("op_2090"), val = tensor([1, 2, 1, 1])]; + tensor x_37_cast_fp16 = tile(reps = var_2090, x = x_35_cast_fp16)[name = string("x_37_cast_fp16")]; + tensor var_2102 = const()[name = string("op_2102"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_11_cast_fp16 = reshape(shape = var_2102, x = x_37_cast_fp16)[name = string("key_states_11_cast_fp16")]; + tensor x_41_axes_0 = const()[name = string("x_41_axes_0"), val = tensor([1])]; + tensor x_41_cast_fp16 = expand_dims(axes = x_41_axes_0, x = V_layer_cache_5_cast_fp16)[name = string("x_41_cast_fp16")]; + tensor var_2110 = const()[name = string("op_2110"), val = tensor([1, 2, 1, 1])]; + tensor x_43_cast_fp16 = tile(reps = var_2110, x = x_41_cast_fp16)[name = string("x_43_cast_fp16")]; + tensor var_2122 = const()[name = string("op_2122"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_15_cast_fp16 = reshape(shape = var_2122, x = x_43_cast_fp16)[name = string("value_states_15_cast_fp16")]; + bool var_2137_transpose_x_1 = const()[name = string("op_2137_transpose_x_1"), val = bool(false)]; + bool var_2137_transpose_y_1 = const()[name = string("op_2137_transpose_y_1"), val = bool(true)]; + tensor var_2137 = matmul(transpose_x = var_2137_transpose_x_1, transpose_y = var_2137_transpose_y_1, x = query_states_9, y = key_states_11_cast_fp16)[name = string("op_2137")]; + fp16 var_2138_to_fp16 = const()[name = string("op_2138_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_13_cast_fp16 = mul(x = var_2137, y = var_2138_to_fp16)[name = string("attn_weights_13_cast_fp16")]; + tensor attn_weights_15_cast_fp16 = add(x = attn_weights_13_cast_fp16, y = causal_mask)[name = string("attn_weights_15_cast_fp16")]; + int32 var_2173 = const()[name = string("op_2173"), val = int32(-1)]; + tensor attn_weights_17_cast_fp16 = softmax(axis = var_2173, x = attn_weights_15_cast_fp16)[name = string("attn_weights_17_cast_fp16")]; + bool attn_output_21_transpose_x_0 = const()[name = string("attn_output_21_transpose_x_0"), val = bool(false)]; + bool attn_output_21_transpose_y_0 = const()[name = string("attn_output_21_transpose_y_0"), val = bool(false)]; + tensor attn_output_21_cast_fp16 = matmul(transpose_x = attn_output_21_transpose_x_0, transpose_y = attn_output_21_transpose_y_0, x = attn_weights_17_cast_fp16, y = value_states_15_cast_fp16)[name = string("attn_output_21_cast_fp16")]; + tensor var_2184_perm_0 = const()[name = string("op_2184_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_2188 = const()[name = string("op_2188"), val = tensor([1, 1, 2048])]; + tensor var_2184_cast_fp16 = transpose(perm = var_2184_perm_0, x = attn_output_21_cast_fp16)[name = string("transpose_70")]; + tensor attn_output_25_cast_fp16 = reshape(shape = var_2188, x = var_2184_cast_fp16)[name = string("attn_output_25_cast_fp16")]; + tensor var_2193 = const()[name = string("op_2193"), val = tensor([0, 2, 1])]; + string var_2209_pad_type_0 = const()[name = string("op_2209_pad_type_0"), val = string("valid")]; + int32 var_2209_groups_0 = const()[name = string("op_2209_groups_0"), val = int32(1)]; + tensor var_2209_strides_0 = const()[name = string("op_2209_strides_0"), val = tensor([1])]; + tensor var_2209_pad_0 = const()[name = string("op_2209_pad_0"), val = tensor([0, 0])]; + tensor var_2209_dilations_0 = const()[name = string("op_2209_dilations_0"), val = tensor([1])]; + tensor squeeze_2_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672171136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676365504))))[name = string("squeeze_2_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_2194_cast_fp16 = transpose(perm = var_2193, x = attn_output_25_cast_fp16)[name = string("transpose_69")]; + tensor var_2209_cast_fp16 = conv(dilations = var_2209_dilations_0, groups = var_2209_groups_0, pad = var_2209_pad_0, pad_type = var_2209_pad_type_0, strides = var_2209_strides_0, weight = squeeze_2_cast_fp16_to_fp32_to_fp16_palettized, x = var_2194_cast_fp16)[name = string("op_2209_cast_fp16")]; + tensor var_2213 = const()[name = string("op_2213"), val = tensor([0, 2, 1])]; + tensor attn_output_29_cast_fp16 = transpose(perm = var_2213, x = var_2209_cast_fp16)[name = string("transpose_68")]; + tensor hidden_states_29_cast_fp16 = add(x = hidden_states_21_cast_fp16, y = attn_output_29_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; + int32 var_2226 = const()[name = string("op_2226"), val = int32(-1)]; + fp16 const_86_promoted_to_fp16 = const()[name = string("const_86_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2228_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_86_promoted_to_fp16)[name = string("op_2228_cast_fp16")]; + bool input_47_interleave_0 = const()[name = string("input_47_interleave_0"), val = bool(false)]; + tensor input_47_cast_fp16 = concat(axis = var_2226, interleave = input_47_interleave_0, values = (hidden_states_29_cast_fp16, var_2228_cast_fp16))[name = string("input_47_cast_fp16")]; + tensor normed_45_axes_0 = const()[name = string("normed_45_axes_0"), val = tensor([-1])]; + fp16 var_2223_to_fp16 = const()[name = string("op_2223_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_45_cast_fp16 = layer_norm(axes = normed_45_axes_0, epsilon = var_2223_to_fp16, x = input_47_cast_fp16)[name = string("normed_45_cast_fp16")]; + tensor normed_47_begin_0 = const()[name = string("normed_47_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_47_end_0 = const()[name = string("normed_47_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_47_end_mask_0 = const()[name = string("normed_47_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_47_cast_fp16 = slice_by_index(begin = normed_47_begin_0, end = normed_47_end_0, end_mask = normed_47_end_mask_0, x = normed_45_cast_fp16)[name = string("normed_47_cast_fp16")]; + tensor const_89_promoted_to_fp16 = const()[name = string("const_89_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676496640)))]; + tensor x_45_cast_fp16 = mul(x = normed_47_cast_fp16, y = const_89_promoted_to_fp16)[name = string("x_45_cast_fp16")]; + tensor var_2253 = const()[name = string("op_2253"), val = tensor([0, 2, 1])]; + tensor input_49_axes_0 = const()[name = string("input_49_axes_0"), val = tensor([2])]; + tensor var_2254 = transpose(perm = var_2253, x = x_45_cast_fp16)[name = string("transpose_67")]; + tensor input_49 = expand_dims(axes = input_49_axes_0, x = var_2254)[name = string("input_49")]; + string input_51_pad_type_0 = const()[name = string("input_51_pad_type_0"), val = string("valid")]; + tensor input_51_strides_0 = const()[name = string("input_51_strides_0"), val = tensor([1, 1])]; + tensor input_51_pad_0 = const()[name = string("input_51_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_51_dilations_0 = const()[name = string("input_51_dilations_0"), val = tensor([1, 1])]; + int32 input_51_groups_0 = const()[name = string("input_51_groups_0"), val = int32(1)]; + tensor input_51 = conv(dilations = input_51_dilations_0, groups = input_51_groups_0, pad = input_51_pad_0, pad_type = input_51_pad_type_0, strides = input_51_strides_0, weight = model_model_layers_16_mlp_gate_proj_weight_palettized, x = input_49)[name = string("input_51")]; + string b_5_pad_type_0 = const()[name = string("b_5_pad_type_0"), val = string("valid")]; + tensor b_5_strides_0 = const()[name = string("b_5_strides_0"), val = tensor([1, 1])]; + tensor b_5_pad_0 = const()[name = string("b_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_5_dilations_0 = const()[name = string("b_5_dilations_0"), val = tensor([1, 1])]; + int32 b_5_groups_0 = const()[name = string("b_5_groups_0"), val = int32(1)]; + tensor b_5 = conv(dilations = b_5_dilations_0, groups = b_5_groups_0, pad = b_5_pad_0, pad_type = b_5_pad_type_0, strides = b_5_strides_0, weight = model_model_layers_16_mlp_up_proj_weight_palettized, x = input_49)[name = string("b_5")]; + tensor c_5 = silu(x = input_51)[name = string("c_5")]; + tensor input_53 = mul(x = c_5, y = b_5)[name = string("input_53")]; + string e_5_pad_type_0 = const()[name = string("e_5_pad_type_0"), val = string("valid")]; + tensor e_5_strides_0 = const()[name = string("e_5_strides_0"), val = tensor([1, 1])]; + tensor e_5_pad_0 = const()[name = string("e_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_5_dilations_0 = const()[name = string("e_5_dilations_0"), val = tensor([1, 1])]; + int32 e_5_groups_0 = const()[name = string("e_5_groups_0"), val = int32(1)]; + tensor e_5 = conv(dilations = e_5_dilations_0, groups = e_5_groups_0, pad = e_5_pad_0, pad_type = e_5_pad_type_0, strides = e_5_strides_0, weight = model_model_layers_16_mlp_down_proj_weight_palettized, x = input_53)[name = string("e_5")]; + tensor var_2276_axes_0 = const()[name = string("op_2276_axes_0"), val = tensor([2])]; + tensor var_2276 = squeeze(axes = var_2276_axes_0, x = e_5)[name = string("op_2276")]; + tensor var_2277 = const()[name = string("op_2277"), val = tensor([0, 2, 1])]; + tensor var_2278 = transpose(perm = var_2277, x = var_2276)[name = string("transpose_66")]; + tensor hidden_states_31_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = var_2278)[name = string("hidden_states_31_cast_fp16")]; + int32 var_2290 = const()[name = string("op_2290"), val = int32(-1)]; + fp16 const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2292_cast_fp16 = mul(x = hidden_states_31_cast_fp16, y = const_90_promoted_to_fp16)[name = string("op_2292_cast_fp16")]; + bool input_55_interleave_0 = const()[name = string("input_55_interleave_0"), val = bool(false)]; + tensor input_55_cast_fp16 = concat(axis = var_2290, interleave = input_55_interleave_0, values = (hidden_states_31_cast_fp16, var_2292_cast_fp16))[name = string("input_55_cast_fp16")]; + tensor normed_49_axes_0 = const()[name = string("normed_49_axes_0"), val = tensor([-1])]; + fp16 var_2287_to_fp16 = const()[name = string("op_2287_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_49_cast_fp16 = layer_norm(axes = normed_49_axes_0, epsilon = var_2287_to_fp16, x = input_55_cast_fp16)[name = string("normed_49_cast_fp16")]; + tensor normed_51_begin_0 = const()[name = string("normed_51_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_51_end_0 = const()[name = string("normed_51_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_51_end_mask_0 = const()[name = string("normed_51_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_51_cast_fp16 = slice_by_index(begin = normed_51_begin_0, end = normed_51_end_0, end_mask = normed_51_end_mask_0, x = normed_49_cast_fp16)[name = string("normed_51_cast_fp16")]; + tensor const_93_promoted_to_fp16 = const()[name = string("const_93_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676500800)))]; + tensor hidden_states_33_cast_fp16 = mul(x = normed_51_cast_fp16, y = const_93_promoted_to_fp16)[name = string("hidden_states_33_cast_fp16")]; + tensor var_2309 = const()[name = string("op_2309"), val = tensor([0, 2, 1])]; + tensor var_2312_axes_0 = const()[name = string("op_2312_axes_0"), val = tensor([2])]; + tensor var_2310_cast_fp16 = transpose(perm = var_2309, x = hidden_states_33_cast_fp16)[name = string("transpose_65")]; + tensor var_2312_cast_fp16 = expand_dims(axes = var_2312_axes_0, x = var_2310_cast_fp16)[name = string("op_2312_cast_fp16")]; + string var_2328_pad_type_0 = const()[name = string("op_2328_pad_type_0"), val = string("valid")]; + tensor var_2328_strides_0 = const()[name = string("op_2328_strides_0"), val = tensor([1, 1])]; + tensor var_2328_pad_0 = const()[name = string("op_2328_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2328_dilations_0 = const()[name = string("op_2328_dilations_0"), val = tensor([1, 1])]; + int32 var_2328_groups_0 = const()[name = string("op_2328_groups_0"), val = int32(1)]; + tensor var_2328 = conv(dilations = var_2328_dilations_0, groups = var_2328_groups_0, pad = var_2328_pad_0, pad_type = var_2328_pad_type_0, strides = var_2328_strides_0, weight = model_model_layers_17_self_attn_q_proj_weight_palettized, x = var_2312_cast_fp16)[name = string("op_2328")]; + tensor var_2333 = const()[name = string("op_2333"), val = tensor([1, 16, 1, 128])]; + tensor var_2334 = reshape(shape = var_2333, x = var_2328)[name = string("op_2334")]; + string var_2350_pad_type_0 = const()[name = string("op_2350_pad_type_0"), val = string("valid")]; + tensor var_2350_strides_0 = const()[name = string("op_2350_strides_0"), val = tensor([1, 1])]; + tensor var_2350_pad_0 = const()[name = string("op_2350_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2350_dilations_0 = const()[name = string("op_2350_dilations_0"), val = tensor([1, 1])]; + int32 var_2350_groups_0 = const()[name = string("op_2350_groups_0"), val = int32(1)]; + tensor var_2350 = conv(dilations = var_2350_dilations_0, groups = var_2350_groups_0, pad = var_2350_pad_0, pad_type = var_2350_pad_type_0, strides = var_2350_strides_0, weight = model_model_layers_17_self_attn_k_proj_weight_palettized, x = var_2312_cast_fp16)[name = string("op_2350")]; + tensor var_2355 = const()[name = string("op_2355"), val = tensor([1, 8, 1, 128])]; + tensor var_2356 = reshape(shape = var_2355, x = var_2350)[name = string("op_2356")]; + string var_2372_pad_type_0 = const()[name = string("op_2372_pad_type_0"), val = string("valid")]; + tensor var_2372_strides_0 = const()[name = string("op_2372_strides_0"), val = tensor([1, 1])]; + tensor var_2372_pad_0 = const()[name = string("op_2372_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2372_dilations_0 = const()[name = string("op_2372_dilations_0"), val = tensor([1, 1])]; + int32 var_2372_groups_0 = const()[name = string("op_2372_groups_0"), val = int32(1)]; + tensor var_2372 = conv(dilations = var_2372_dilations_0, groups = var_2372_groups_0, pad = var_2372_pad_0, pad_type = var_2372_pad_type_0, strides = var_2372_strides_0, weight = model_model_layers_17_self_attn_v_proj_weight_palettized, x = var_2312_cast_fp16)[name = string("op_2372")]; + tensor var_2377 = const()[name = string("op_2377"), val = tensor([1, 8, 1, 128])]; + tensor var_2378 = reshape(shape = var_2377, x = var_2372)[name = string("op_2378")]; + int32 var_2393 = const()[name = string("op_2393"), val = int32(-1)]; + fp16 const_94_promoted = const()[name = string("const_94_promoted"), val = fp16(-0x1p+0)]; + tensor var_2395 = mul(x = var_2334, y = const_94_promoted)[name = string("op_2395")]; + bool input_59_interleave_0 = const()[name = string("input_59_interleave_0"), val = bool(false)]; + tensor input_59 = concat(axis = var_2393, interleave = input_59_interleave_0, values = (var_2334, var_2395))[name = string("input_59")]; + tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; + fp16 var_2390_to_fp16 = const()[name = string("op_2390_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_2390_to_fp16, x = input_59)[name = string("normed_53_cast_fp16")]; + tensor normed_55_begin_0 = const()[name = string("normed_55_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_55_end_0 = const()[name = string("normed_55_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_55_end_mask_0 = const()[name = string("normed_55_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_55 = slice_by_index(begin = normed_55_begin_0, end = normed_55_end_0, end_mask = normed_55_end_mask_0, x = normed_53_cast_fp16)[name = string("normed_55")]; + tensor const_97 = const()[name = string("const_97"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676504960)))]; + tensor q_7 = mul(x = normed_55, y = const_97)[name = string("q_7")]; + int32 var_2418 = const()[name = string("op_2418"), val = int32(-1)]; + fp16 const_98_promoted = const()[name = string("const_98_promoted"), val = fp16(-0x1p+0)]; + tensor var_2420 = mul(x = var_2356, y = const_98_promoted)[name = string("op_2420")]; + bool input_61_interleave_0 = const()[name = string("input_61_interleave_0"), val = bool(false)]; + tensor input_61 = concat(axis = var_2418, interleave = input_61_interleave_0, values = (var_2356, var_2420))[name = string("input_61")]; + tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; + fp16 var_2415_to_fp16 = const()[name = string("op_2415_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_2415_to_fp16, x = input_61)[name = string("normed_57_cast_fp16")]; + tensor normed_59_begin_0 = const()[name = string("normed_59_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_59_end_0 = const()[name = string("normed_59_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_59_end_mask_0 = const()[name = string("normed_59_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_59 = slice_by_index(begin = normed_59_begin_0, end = normed_59_end_0, end_mask = normed_59_end_mask_0, x = normed_57_cast_fp16)[name = string("normed_59")]; + tensor const_101 = const()[name = string("const_101"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676505280)))]; + tensor k_7 = mul(x = normed_59, y = const_101)[name = string("k_7")]; + tensor var_2434 = mul(x = q_7, y = cos_1_cast_fp16)[name = string("op_2434")]; + tensor x1_13_begin_0 = const()[name = string("x1_13_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_13_end_0 = const()[name = string("x1_13_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_13_end_mask_0 = const()[name = string("x1_13_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_13 = slice_by_index(begin = x1_13_begin_0, end = x1_13_end_0, end_mask = x1_13_end_mask_0, x = q_7)[name = string("x1_13")]; + tensor x2_13_begin_0 = const()[name = string("x2_13_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_13_end_0 = const()[name = string("x2_13_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_13_end_mask_0 = const()[name = string("x2_13_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_13 = slice_by_index(begin = x2_13_begin_0, end = x2_13_end_0, end_mask = x2_13_end_mask_0, x = q_7)[name = string("x2_13")]; + fp16 const_104_promoted = const()[name = string("const_104_promoted"), val = fp16(-0x1p+0)]; + tensor var_2455 = mul(x = x2_13, y = const_104_promoted)[name = string("op_2455")]; + int32 var_2457 = const()[name = string("op_2457"), val = int32(-1)]; + bool var_2458_interleave_0 = const()[name = string("op_2458_interleave_0"), val = bool(false)]; + tensor var_2458 = concat(axis = var_2457, interleave = var_2458_interleave_0, values = (var_2455, x1_13))[name = string("op_2458")]; + tensor var_2459 = mul(x = var_2458, y = sin_1_cast_fp16)[name = string("op_2459")]; + tensor query_states_13 = add(x = var_2434, y = var_2459)[name = string("query_states_13")]; + tensor var_2462 = mul(x = k_7, y = cos_1_cast_fp16)[name = string("op_2462")]; + tensor x1_15_begin_0 = const()[name = string("x1_15_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_15_end_0 = const()[name = string("x1_15_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_15_end_mask_0 = const()[name = string("x1_15_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_15 = slice_by_index(begin = x1_15_begin_0, end = x1_15_end_0, end_mask = x1_15_end_mask_0, x = k_7)[name = string("x1_15")]; + tensor x2_15_begin_0 = const()[name = string("x2_15_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_15_end_0 = const()[name = string("x2_15_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_15_end_mask_0 = const()[name = string("x2_15_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_15 = slice_by_index(begin = x2_15_begin_0, end = x2_15_end_0, end_mask = x2_15_end_mask_0, x = k_7)[name = string("x2_15")]; + fp16 const_107_promoted = const()[name = string("const_107_promoted"), val = fp16(-0x1p+0)]; + tensor var_2483 = mul(x = x2_15, y = const_107_promoted)[name = string("op_2483")]; + int32 var_2485 = const()[name = string("op_2485"), val = int32(-1)]; + bool var_2486_interleave_0 = const()[name = string("op_2486_interleave_0"), val = bool(false)]; + tensor var_2486 = concat(axis = var_2485, interleave = var_2486_interleave_0, values = (var_2483, x1_15))[name = string("op_2486")]; + tensor var_2487 = mul(x = var_2486, y = sin_1_cast_fp16)[name = string("op_2487")]; + tensor key_states_13 = add(x = var_2462, y = var_2487)[name = string("key_states_13")]; + tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([17])]; + tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; + tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; + tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([18])]; + int32 concat_26_axis_0 = const()[name = string("concat_26_axis_0"), val = int32(0)]; + bool concat_26_interleave_0 = const()[name = string("concat_26_interleave_0"), val = bool(false)]; + tensor concat_26 = concat(axis = concat_26_axis_0, interleave = concat_26_interleave_0, values = (expand_dims_36, expand_dims_37, current_pos, expand_dims_39))[name = string("concat_26")]; + tensor concat_27_values1_0 = const()[name = string("concat_27_values1_0"), val = tensor([0])]; + tensor concat_27_values3_0 = const()[name = string("concat_27_values3_0"), val = tensor([0])]; + int32 concat_27_axis_0 = const()[name = string("concat_27_axis_0"), val = int32(0)]; + bool concat_27_interleave_0 = const()[name = string("concat_27_interleave_0"), val = bool(false)]; + tensor concat_27 = concat(axis = concat_27_axis_0, interleave = concat_27_interleave_0, values = (expand_dims_40, concat_27_values1_0, var_1004, concat_27_values3_0))[name = string("concat_27")]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_26, begin_mask = model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0, end = concat_27, end_mask = model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_7_stride_0, update = key_states_13, x = coreml_update_state_33)[name = string("model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_62_write_state")]; + tensor coreml_update_state_34 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_62")]; + tensor expand_dims_42 = const()[name = string("expand_dims_42"), val = tensor([45])]; + tensor expand_dims_43 = const()[name = string("expand_dims_43"), val = tensor([0])]; + tensor expand_dims_45 = const()[name = string("expand_dims_45"), val = tensor([0])]; + tensor expand_dims_46 = const()[name = string("expand_dims_46"), val = tensor([46])]; + int32 concat_30_axis_0 = const()[name = string("concat_30_axis_0"), val = int32(0)]; + bool concat_30_interleave_0 = const()[name = string("concat_30_interleave_0"), val = bool(false)]; + tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (expand_dims_42, expand_dims_43, current_pos, expand_dims_45))[name = string("concat_30")]; + tensor concat_31_values1_0 = const()[name = string("concat_31_values1_0"), val = tensor([0])]; + tensor concat_31_values3_0 = const()[name = string("concat_31_values3_0"), val = tensor([0])]; + int32 concat_31_axis_0 = const()[name = string("concat_31_axis_0"), val = int32(0)]; + bool concat_31_interleave_0 = const()[name = string("concat_31_interleave_0"), val = bool(false)]; + tensor concat_31 = concat(axis = concat_31_axis_0, interleave = concat_31_interleave_0, values = (expand_dims_46, concat_31_values1_0, var_1004, concat_31_values3_0))[name = string("concat_31")]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_30, begin_mask = model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0, end = concat_31, end_mask = model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_8_stride_0, update = var_2378, x = coreml_update_state_34)[name = string("model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_63_write_state")]; + tensor coreml_update_state_35 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_63")]; + tensor var_2542_begin_0 = const()[name = string("op_2542_begin_0"), val = tensor([17, 0, 0, 0])]; + tensor var_2542_end_0 = const()[name = string("op_2542_end_0"), val = tensor([18, 8, 1024, 128])]; + tensor var_2542_end_mask_0 = const()[name = string("op_2542_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2542_cast_fp16 = slice_by_index(begin = var_2542_begin_0, end = var_2542_end_0, end_mask = var_2542_end_mask_0, x = coreml_update_state_35)[name = string("op_2542_cast_fp16")]; + tensor K_layer_cache_7_axes_0 = const()[name = string("K_layer_cache_7_axes_0"), val = tensor([0])]; + tensor K_layer_cache_7_cast_fp16 = squeeze(axes = K_layer_cache_7_axes_0, x = var_2542_cast_fp16)[name = string("K_layer_cache_7_cast_fp16")]; + tensor var_2549_begin_0 = const()[name = string("op_2549_begin_0"), val = tensor([45, 0, 0, 0])]; + tensor var_2549_end_0 = const()[name = string("op_2549_end_0"), val = tensor([46, 8, 1024, 128])]; + tensor var_2549_end_mask_0 = const()[name = string("op_2549_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2549_cast_fp16 = slice_by_index(begin = var_2549_begin_0, end = var_2549_end_0, end_mask = var_2549_end_mask_0, x = coreml_update_state_35)[name = string("op_2549_cast_fp16")]; + tensor V_layer_cache_7_axes_0 = const()[name = string("V_layer_cache_7_axes_0"), val = tensor([0])]; + tensor V_layer_cache_7_cast_fp16 = squeeze(axes = V_layer_cache_7_axes_0, x = var_2549_cast_fp16)[name = string("V_layer_cache_7_cast_fp16")]; + tensor x_51_axes_0 = const()[name = string("x_51_axes_0"), val = tensor([1])]; + tensor x_51_cast_fp16 = expand_dims(axes = x_51_axes_0, x = K_layer_cache_7_cast_fp16)[name = string("x_51_cast_fp16")]; + tensor var_2586 = const()[name = string("op_2586"), val = tensor([1, 2, 1, 1])]; + tensor x_53_cast_fp16 = tile(reps = var_2586, x = x_51_cast_fp16)[name = string("x_53_cast_fp16")]; + tensor var_2598 = const()[name = string("op_2598"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_15_cast_fp16 = reshape(shape = var_2598, x = x_53_cast_fp16)[name = string("key_states_15_cast_fp16")]; + tensor x_57_axes_0 = const()[name = string("x_57_axes_0"), val = tensor([1])]; + tensor x_57_cast_fp16 = expand_dims(axes = x_57_axes_0, x = V_layer_cache_7_cast_fp16)[name = string("x_57_cast_fp16")]; + tensor var_2606 = const()[name = string("op_2606"), val = tensor([1, 2, 1, 1])]; + tensor x_59_cast_fp16 = tile(reps = var_2606, x = x_57_cast_fp16)[name = string("x_59_cast_fp16")]; + tensor var_2618 = const()[name = string("op_2618"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_21_cast_fp16 = reshape(shape = var_2618, x = x_59_cast_fp16)[name = string("value_states_21_cast_fp16")]; + bool var_2633_transpose_x_1 = const()[name = string("op_2633_transpose_x_1"), val = bool(false)]; + bool var_2633_transpose_y_1 = const()[name = string("op_2633_transpose_y_1"), val = bool(true)]; + tensor var_2633 = matmul(transpose_x = var_2633_transpose_x_1, transpose_y = var_2633_transpose_y_1, x = query_states_13, y = key_states_15_cast_fp16)[name = string("op_2633")]; + fp16 var_2634_to_fp16 = const()[name = string("op_2634_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_19_cast_fp16 = mul(x = var_2633, y = var_2634_to_fp16)[name = string("attn_weights_19_cast_fp16")]; + tensor attn_weights_21_cast_fp16 = add(x = attn_weights_19_cast_fp16, y = causal_mask)[name = string("attn_weights_21_cast_fp16")]; + int32 var_2669 = const()[name = string("op_2669"), val = int32(-1)]; + tensor attn_weights_23_cast_fp16 = softmax(axis = var_2669, x = attn_weights_21_cast_fp16)[name = string("attn_weights_23_cast_fp16")]; + bool attn_output_31_transpose_x_0 = const()[name = string("attn_output_31_transpose_x_0"), val = bool(false)]; + bool attn_output_31_transpose_y_0 = const()[name = string("attn_output_31_transpose_y_0"), val = bool(false)]; + tensor attn_output_31_cast_fp16 = matmul(transpose_x = attn_output_31_transpose_x_0, transpose_y = attn_output_31_transpose_y_0, x = attn_weights_23_cast_fp16, y = value_states_21_cast_fp16)[name = string("attn_output_31_cast_fp16")]; + tensor var_2680_perm_0 = const()[name = string("op_2680_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_2684 = const()[name = string("op_2684"), val = tensor([1, 1, 2048])]; + tensor var_2680_cast_fp16 = transpose(perm = var_2680_perm_0, x = attn_output_31_cast_fp16)[name = string("transpose_64")]; + tensor attn_output_35_cast_fp16 = reshape(shape = var_2684, x = var_2680_cast_fp16)[name = string("attn_output_35_cast_fp16")]; + tensor var_2689 = const()[name = string("op_2689"), val = tensor([0, 2, 1])]; + string var_2705_pad_type_0 = const()[name = string("op_2705_pad_type_0"), val = string("valid")]; + int32 var_2705_groups_0 = const()[name = string("op_2705_groups_0"), val = int32(1)]; + tensor var_2705_strides_0 = const()[name = string("op_2705_strides_0"), val = tensor([1])]; + tensor var_2705_pad_0 = const()[name = string("op_2705_pad_0"), val = tensor([0, 0])]; + tensor var_2705_dilations_0 = const()[name = string("op_2705_dilations_0"), val = tensor([1])]; + tensor squeeze_3_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676505600))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680699968))))[name = string("squeeze_3_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_2690_cast_fp16 = transpose(perm = var_2689, x = attn_output_35_cast_fp16)[name = string("transpose_63")]; + tensor var_2705_cast_fp16 = conv(dilations = var_2705_dilations_0, groups = var_2705_groups_0, pad = var_2705_pad_0, pad_type = var_2705_pad_type_0, strides = var_2705_strides_0, weight = squeeze_3_cast_fp16_to_fp32_to_fp16_palettized, x = var_2690_cast_fp16)[name = string("op_2705_cast_fp16")]; + tensor var_2709 = const()[name = string("op_2709"), val = tensor([0, 2, 1])]; + tensor attn_output_39_cast_fp16 = transpose(perm = var_2709, x = var_2705_cast_fp16)[name = string("transpose_62")]; + tensor hidden_states_39_cast_fp16 = add(x = hidden_states_31_cast_fp16, y = attn_output_39_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; + int32 var_2722 = const()[name = string("op_2722"), val = int32(-1)]; + fp16 const_116_promoted_to_fp16 = const()[name = string("const_116_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2724_cast_fp16 = mul(x = hidden_states_39_cast_fp16, y = const_116_promoted_to_fp16)[name = string("op_2724_cast_fp16")]; + bool input_65_interleave_0 = const()[name = string("input_65_interleave_0"), val = bool(false)]; + tensor input_65_cast_fp16 = concat(axis = var_2722, interleave = input_65_interleave_0, values = (hidden_states_39_cast_fp16, var_2724_cast_fp16))[name = string("input_65_cast_fp16")]; + tensor normed_61_axes_0 = const()[name = string("normed_61_axes_0"), val = tensor([-1])]; + fp16 var_2719_to_fp16 = const()[name = string("op_2719_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_61_cast_fp16 = layer_norm(axes = normed_61_axes_0, epsilon = var_2719_to_fp16, x = input_65_cast_fp16)[name = string("normed_61_cast_fp16")]; + tensor normed_63_begin_0 = const()[name = string("normed_63_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_63_end_0 = const()[name = string("normed_63_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_63_end_mask_0 = const()[name = string("normed_63_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_63_cast_fp16 = slice_by_index(begin = normed_63_begin_0, end = normed_63_end_0, end_mask = normed_63_end_mask_0, x = normed_61_cast_fp16)[name = string("normed_63_cast_fp16")]; + tensor const_119_promoted_to_fp16 = const()[name = string("const_119_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680831104)))]; + tensor x_61_cast_fp16 = mul(x = normed_63_cast_fp16, y = const_119_promoted_to_fp16)[name = string("x_61_cast_fp16")]; + tensor var_2749 = const()[name = string("op_2749"), val = tensor([0, 2, 1])]; + tensor input_67_axes_0 = const()[name = string("input_67_axes_0"), val = tensor([2])]; + tensor var_2750 = transpose(perm = var_2749, x = x_61_cast_fp16)[name = string("transpose_61")]; + tensor input_67 = expand_dims(axes = input_67_axes_0, x = var_2750)[name = string("input_67")]; + string input_69_pad_type_0 = const()[name = string("input_69_pad_type_0"), val = string("valid")]; + tensor input_69_strides_0 = const()[name = string("input_69_strides_0"), val = tensor([1, 1])]; + tensor input_69_pad_0 = const()[name = string("input_69_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_69_dilations_0 = const()[name = string("input_69_dilations_0"), val = tensor([1, 1])]; + int32 input_69_groups_0 = const()[name = string("input_69_groups_0"), val = int32(1)]; + tensor input_69 = conv(dilations = input_69_dilations_0, groups = input_69_groups_0, pad = input_69_pad_0, pad_type = input_69_pad_type_0, strides = input_69_strides_0, weight = model_model_layers_17_mlp_gate_proj_weight_palettized, x = input_67)[name = string("input_69")]; + string b_7_pad_type_0 = const()[name = string("b_7_pad_type_0"), val = string("valid")]; + tensor b_7_strides_0 = const()[name = string("b_7_strides_0"), val = tensor([1, 1])]; + tensor b_7_pad_0 = const()[name = string("b_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_7_dilations_0 = const()[name = string("b_7_dilations_0"), val = tensor([1, 1])]; + int32 b_7_groups_0 = const()[name = string("b_7_groups_0"), val = int32(1)]; + tensor b_7 = conv(dilations = b_7_dilations_0, groups = b_7_groups_0, pad = b_7_pad_0, pad_type = b_7_pad_type_0, strides = b_7_strides_0, weight = model_model_layers_17_mlp_up_proj_weight_palettized, x = input_67)[name = string("b_7")]; + tensor c_7 = silu(x = input_69)[name = string("c_7")]; + tensor input_71 = mul(x = c_7, y = b_7)[name = string("input_71")]; + string e_7_pad_type_0 = const()[name = string("e_7_pad_type_0"), val = string("valid")]; + tensor e_7_strides_0 = const()[name = string("e_7_strides_0"), val = tensor([1, 1])]; + tensor e_7_pad_0 = const()[name = string("e_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_7_dilations_0 = const()[name = string("e_7_dilations_0"), val = tensor([1, 1])]; + int32 e_7_groups_0 = const()[name = string("e_7_groups_0"), val = int32(1)]; + tensor e_7 = conv(dilations = e_7_dilations_0, groups = e_7_groups_0, pad = e_7_pad_0, pad_type = e_7_pad_type_0, strides = e_7_strides_0, weight = model_model_layers_17_mlp_down_proj_weight_palettized, x = input_71)[name = string("e_7")]; + tensor var_2772_axes_0 = const()[name = string("op_2772_axes_0"), val = tensor([2])]; + tensor var_2772 = squeeze(axes = var_2772_axes_0, x = e_7)[name = string("op_2772")]; + tensor var_2773 = const()[name = string("op_2773"), val = tensor([0, 2, 1])]; + tensor var_2774 = transpose(perm = var_2773, x = var_2772)[name = string("transpose_60")]; + tensor hidden_states_41_cast_fp16 = add(x = hidden_states_39_cast_fp16, y = var_2774)[name = string("hidden_states_41_cast_fp16")]; + int32 var_2786 = const()[name = string("op_2786"), val = int32(-1)]; + fp16 const_120_promoted_to_fp16 = const()[name = string("const_120_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2788_cast_fp16 = mul(x = hidden_states_41_cast_fp16, y = const_120_promoted_to_fp16)[name = string("op_2788_cast_fp16")]; + bool input_73_interleave_0 = const()[name = string("input_73_interleave_0"), val = bool(false)]; + tensor input_73_cast_fp16 = concat(axis = var_2786, interleave = input_73_interleave_0, values = (hidden_states_41_cast_fp16, var_2788_cast_fp16))[name = string("input_73_cast_fp16")]; + tensor normed_65_axes_0 = const()[name = string("normed_65_axes_0"), val = tensor([-1])]; + fp16 var_2783_to_fp16 = const()[name = string("op_2783_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_65_cast_fp16 = layer_norm(axes = normed_65_axes_0, epsilon = var_2783_to_fp16, x = input_73_cast_fp16)[name = string("normed_65_cast_fp16")]; + tensor normed_67_begin_0 = const()[name = string("normed_67_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_67_end_0 = const()[name = string("normed_67_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_67_end_mask_0 = const()[name = string("normed_67_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_67_cast_fp16 = slice_by_index(begin = normed_67_begin_0, end = normed_67_end_0, end_mask = normed_67_end_mask_0, x = normed_65_cast_fp16)[name = string("normed_67_cast_fp16")]; + tensor const_123_promoted_to_fp16 = const()[name = string("const_123_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680835264)))]; + tensor hidden_states_43_cast_fp16 = mul(x = normed_67_cast_fp16, y = const_123_promoted_to_fp16)[name = string("hidden_states_43_cast_fp16")]; + tensor var_2805 = const()[name = string("op_2805"), val = tensor([0, 2, 1])]; + tensor var_2808_axes_0 = const()[name = string("op_2808_axes_0"), val = tensor([2])]; + tensor var_2806_cast_fp16 = transpose(perm = var_2805, x = hidden_states_43_cast_fp16)[name = string("transpose_59")]; + tensor var_2808_cast_fp16 = expand_dims(axes = var_2808_axes_0, x = var_2806_cast_fp16)[name = string("op_2808_cast_fp16")]; + string var_2824_pad_type_0 = const()[name = string("op_2824_pad_type_0"), val = string("valid")]; + tensor var_2824_strides_0 = const()[name = string("op_2824_strides_0"), val = tensor([1, 1])]; + tensor var_2824_pad_0 = const()[name = string("op_2824_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2824_dilations_0 = const()[name = string("op_2824_dilations_0"), val = tensor([1, 1])]; + int32 var_2824_groups_0 = const()[name = string("op_2824_groups_0"), val = int32(1)]; + tensor var_2824 = conv(dilations = var_2824_dilations_0, groups = var_2824_groups_0, pad = var_2824_pad_0, pad_type = var_2824_pad_type_0, strides = var_2824_strides_0, weight = model_model_layers_18_self_attn_q_proj_weight_palettized, x = var_2808_cast_fp16)[name = string("op_2824")]; + tensor var_2829 = const()[name = string("op_2829"), val = tensor([1, 16, 1, 128])]; + tensor var_2830 = reshape(shape = var_2829, x = var_2824)[name = string("op_2830")]; + string var_2846_pad_type_0 = const()[name = string("op_2846_pad_type_0"), val = string("valid")]; + tensor var_2846_strides_0 = const()[name = string("op_2846_strides_0"), val = tensor([1, 1])]; + tensor var_2846_pad_0 = const()[name = string("op_2846_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2846_dilations_0 = const()[name = string("op_2846_dilations_0"), val = tensor([1, 1])]; + int32 var_2846_groups_0 = const()[name = string("op_2846_groups_0"), val = int32(1)]; + tensor var_2846 = conv(dilations = var_2846_dilations_0, groups = var_2846_groups_0, pad = var_2846_pad_0, pad_type = var_2846_pad_type_0, strides = var_2846_strides_0, weight = model_model_layers_18_self_attn_k_proj_weight_palettized, x = var_2808_cast_fp16)[name = string("op_2846")]; + tensor var_2851 = const()[name = string("op_2851"), val = tensor([1, 8, 1, 128])]; + tensor var_2852 = reshape(shape = var_2851, x = var_2846)[name = string("op_2852")]; + string var_2868_pad_type_0 = const()[name = string("op_2868_pad_type_0"), val = string("valid")]; + tensor var_2868_strides_0 = const()[name = string("op_2868_strides_0"), val = tensor([1, 1])]; + tensor var_2868_pad_0 = const()[name = string("op_2868_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_2868_dilations_0 = const()[name = string("op_2868_dilations_0"), val = tensor([1, 1])]; + int32 var_2868_groups_0 = const()[name = string("op_2868_groups_0"), val = int32(1)]; + tensor var_2868 = conv(dilations = var_2868_dilations_0, groups = var_2868_groups_0, pad = var_2868_pad_0, pad_type = var_2868_pad_type_0, strides = var_2868_strides_0, weight = model_model_layers_18_self_attn_v_proj_weight_palettized, x = var_2808_cast_fp16)[name = string("op_2868")]; + tensor var_2873 = const()[name = string("op_2873"), val = tensor([1, 8, 1, 128])]; + tensor var_2874 = reshape(shape = var_2873, x = var_2868)[name = string("op_2874")]; + int32 var_2889 = const()[name = string("op_2889"), val = int32(-1)]; + fp16 const_124_promoted = const()[name = string("const_124_promoted"), val = fp16(-0x1p+0)]; + tensor var_2891 = mul(x = var_2830, y = const_124_promoted)[name = string("op_2891")]; + bool input_77_interleave_0 = const()[name = string("input_77_interleave_0"), val = bool(false)]; + tensor input_77 = concat(axis = var_2889, interleave = input_77_interleave_0, values = (var_2830, var_2891))[name = string("input_77")]; + tensor normed_69_axes_0 = const()[name = string("normed_69_axes_0"), val = tensor([-1])]; + fp16 var_2886_to_fp16 = const()[name = string("op_2886_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_69_cast_fp16 = layer_norm(axes = normed_69_axes_0, epsilon = var_2886_to_fp16, x = input_77)[name = string("normed_69_cast_fp16")]; + tensor normed_71_begin_0 = const()[name = string("normed_71_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_71_end_0 = const()[name = string("normed_71_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_71_end_mask_0 = const()[name = string("normed_71_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_71 = slice_by_index(begin = normed_71_begin_0, end = normed_71_end_0, end_mask = normed_71_end_mask_0, x = normed_69_cast_fp16)[name = string("normed_71")]; + tensor const_127 = const()[name = string("const_127"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680839424)))]; + tensor q_9 = mul(x = normed_71, y = const_127)[name = string("q_9")]; + int32 var_2914 = const()[name = string("op_2914"), val = int32(-1)]; + fp16 const_128_promoted = const()[name = string("const_128_promoted"), val = fp16(-0x1p+0)]; + tensor var_2916 = mul(x = var_2852, y = const_128_promoted)[name = string("op_2916")]; + bool input_79_interleave_0 = const()[name = string("input_79_interleave_0"), val = bool(false)]; + tensor input_79 = concat(axis = var_2914, interleave = input_79_interleave_0, values = (var_2852, var_2916))[name = string("input_79")]; + tensor normed_73_axes_0 = const()[name = string("normed_73_axes_0"), val = tensor([-1])]; + fp16 var_2911_to_fp16 = const()[name = string("op_2911_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_73_cast_fp16 = layer_norm(axes = normed_73_axes_0, epsilon = var_2911_to_fp16, x = input_79)[name = string("normed_73_cast_fp16")]; + tensor normed_75_begin_0 = const()[name = string("normed_75_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_75_end_0 = const()[name = string("normed_75_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_75_end_mask_0 = const()[name = string("normed_75_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_75 = slice_by_index(begin = normed_75_begin_0, end = normed_75_end_0, end_mask = normed_75_end_mask_0, x = normed_73_cast_fp16)[name = string("normed_75")]; + tensor const_131 = const()[name = string("const_131"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680839744)))]; + tensor k_9 = mul(x = normed_75, y = const_131)[name = string("k_9")]; + tensor var_2930 = mul(x = q_9, y = cos_1_cast_fp16)[name = string("op_2930")]; + tensor x1_17_begin_0 = const()[name = string("x1_17_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_17_end_0 = const()[name = string("x1_17_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_17_end_mask_0 = const()[name = string("x1_17_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_17 = slice_by_index(begin = x1_17_begin_0, end = x1_17_end_0, end_mask = x1_17_end_mask_0, x = q_9)[name = string("x1_17")]; + tensor x2_17_begin_0 = const()[name = string("x2_17_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_17_end_0 = const()[name = string("x2_17_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_17_end_mask_0 = const()[name = string("x2_17_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_17 = slice_by_index(begin = x2_17_begin_0, end = x2_17_end_0, end_mask = x2_17_end_mask_0, x = q_9)[name = string("x2_17")]; + fp16 const_134_promoted = const()[name = string("const_134_promoted"), val = fp16(-0x1p+0)]; + tensor var_2951 = mul(x = x2_17, y = const_134_promoted)[name = string("op_2951")]; + int32 var_2953 = const()[name = string("op_2953"), val = int32(-1)]; + bool var_2954_interleave_0 = const()[name = string("op_2954_interleave_0"), val = bool(false)]; + tensor var_2954 = concat(axis = var_2953, interleave = var_2954_interleave_0, values = (var_2951, x1_17))[name = string("op_2954")]; + tensor var_2955 = mul(x = var_2954, y = sin_1_cast_fp16)[name = string("op_2955")]; + tensor query_states_17 = add(x = var_2930, y = var_2955)[name = string("query_states_17")]; + tensor var_2958 = mul(x = k_9, y = cos_1_cast_fp16)[name = string("op_2958")]; + tensor x1_19_begin_0 = const()[name = string("x1_19_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_19_end_0 = const()[name = string("x1_19_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_19_end_mask_0 = const()[name = string("x1_19_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_19 = slice_by_index(begin = x1_19_begin_0, end = x1_19_end_0, end_mask = x1_19_end_mask_0, x = k_9)[name = string("x1_19")]; + tensor x2_19_begin_0 = const()[name = string("x2_19_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_19_end_0 = const()[name = string("x2_19_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_19_end_mask_0 = const()[name = string("x2_19_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_19 = slice_by_index(begin = x2_19_begin_0, end = x2_19_end_0, end_mask = x2_19_end_mask_0, x = k_9)[name = string("x2_19")]; + fp16 const_137_promoted = const()[name = string("const_137_promoted"), val = fp16(-0x1p+0)]; + tensor var_2979 = mul(x = x2_19, y = const_137_promoted)[name = string("op_2979")]; + int32 var_2981 = const()[name = string("op_2981"), val = int32(-1)]; + bool var_2982_interleave_0 = const()[name = string("op_2982_interleave_0"), val = bool(false)]; + tensor var_2982 = concat(axis = var_2981, interleave = var_2982_interleave_0, values = (var_2979, x1_19))[name = string("op_2982")]; + tensor var_2983 = mul(x = var_2982, y = sin_1_cast_fp16)[name = string("op_2983")]; + tensor key_states_17 = add(x = var_2958, y = var_2983)[name = string("key_states_17")]; + tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([18])]; + tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; + tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; + tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([19])]; + int32 concat_34_axis_0 = const()[name = string("concat_34_axis_0"), val = int32(0)]; + bool concat_34_interleave_0 = const()[name = string("concat_34_interleave_0"), val = bool(false)]; + tensor concat_34 = concat(axis = concat_34_axis_0, interleave = concat_34_interleave_0, values = (expand_dims_48, expand_dims_49, current_pos, expand_dims_51))[name = string("concat_34")]; + tensor concat_35_values1_0 = const()[name = string("concat_35_values1_0"), val = tensor([0])]; + tensor concat_35_values3_0 = const()[name = string("concat_35_values3_0"), val = tensor([0])]; + int32 concat_35_axis_0 = const()[name = string("concat_35_axis_0"), val = int32(0)]; + bool concat_35_interleave_0 = const()[name = string("concat_35_interleave_0"), val = bool(false)]; + tensor concat_35 = concat(axis = concat_35_axis_0, interleave = concat_35_interleave_0, values = (expand_dims_52, concat_35_values1_0, var_1004, concat_35_values3_0))[name = string("concat_35")]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_34, begin_mask = model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0, end = concat_35, end_mask = model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_9_stride_0, update = key_states_17, x = coreml_update_state_35)[name = string("model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_64_write_state")]; + tensor coreml_update_state_36 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_64")]; + tensor expand_dims_54 = const()[name = string("expand_dims_54"), val = tensor([46])]; + tensor expand_dims_55 = const()[name = string("expand_dims_55"), val = tensor([0])]; + tensor expand_dims_57 = const()[name = string("expand_dims_57"), val = tensor([0])]; + tensor expand_dims_58 = const()[name = string("expand_dims_58"), val = tensor([47])]; + int32 concat_38_axis_0 = const()[name = string("concat_38_axis_0"), val = int32(0)]; + bool concat_38_interleave_0 = const()[name = string("concat_38_interleave_0"), val = bool(false)]; + tensor concat_38 = concat(axis = concat_38_axis_0, interleave = concat_38_interleave_0, values = (expand_dims_54, expand_dims_55, current_pos, expand_dims_57))[name = string("concat_38")]; + tensor concat_39_values1_0 = const()[name = string("concat_39_values1_0"), val = tensor([0])]; + tensor concat_39_values3_0 = const()[name = string("concat_39_values3_0"), val = tensor([0])]; + int32 concat_39_axis_0 = const()[name = string("concat_39_axis_0"), val = int32(0)]; + bool concat_39_interleave_0 = const()[name = string("concat_39_interleave_0"), val = bool(false)]; + tensor concat_39 = concat(axis = concat_39_axis_0, interleave = concat_39_interleave_0, values = (expand_dims_58, concat_39_values1_0, var_1004, concat_39_values3_0))[name = string("concat_39")]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_38, begin_mask = model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0, end = concat_39, end_mask = model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_10_stride_0, update = var_2874, x = coreml_update_state_36)[name = string("model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_65_write_state")]; + tensor coreml_update_state_37 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_65")]; + tensor var_3038_begin_0 = const()[name = string("op_3038_begin_0"), val = tensor([18, 0, 0, 0])]; + tensor var_3038_end_0 = const()[name = string("op_3038_end_0"), val = tensor([19, 8, 1024, 128])]; + tensor var_3038_end_mask_0 = const()[name = string("op_3038_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3038_cast_fp16 = slice_by_index(begin = var_3038_begin_0, end = var_3038_end_0, end_mask = var_3038_end_mask_0, x = coreml_update_state_37)[name = string("op_3038_cast_fp16")]; + tensor K_layer_cache_9_axes_0 = const()[name = string("K_layer_cache_9_axes_0"), val = tensor([0])]; + tensor K_layer_cache_9_cast_fp16 = squeeze(axes = K_layer_cache_9_axes_0, x = var_3038_cast_fp16)[name = string("K_layer_cache_9_cast_fp16")]; + tensor var_3045_begin_0 = const()[name = string("op_3045_begin_0"), val = tensor([46, 0, 0, 0])]; + tensor var_3045_end_0 = const()[name = string("op_3045_end_0"), val = tensor([47, 8, 1024, 128])]; + tensor var_3045_end_mask_0 = const()[name = string("op_3045_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3045_cast_fp16 = slice_by_index(begin = var_3045_begin_0, end = var_3045_end_0, end_mask = var_3045_end_mask_0, x = coreml_update_state_37)[name = string("op_3045_cast_fp16")]; + tensor V_layer_cache_9_axes_0 = const()[name = string("V_layer_cache_9_axes_0"), val = tensor([0])]; + tensor V_layer_cache_9_cast_fp16 = squeeze(axes = V_layer_cache_9_axes_0, x = var_3045_cast_fp16)[name = string("V_layer_cache_9_cast_fp16")]; + tensor x_67_axes_0 = const()[name = string("x_67_axes_0"), val = tensor([1])]; + tensor x_67_cast_fp16 = expand_dims(axes = x_67_axes_0, x = K_layer_cache_9_cast_fp16)[name = string("x_67_cast_fp16")]; + tensor var_3082 = const()[name = string("op_3082"), val = tensor([1, 2, 1, 1])]; + tensor x_69_cast_fp16 = tile(reps = var_3082, x = x_67_cast_fp16)[name = string("x_69_cast_fp16")]; + tensor var_3094 = const()[name = string("op_3094"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_19_cast_fp16 = reshape(shape = var_3094, x = x_69_cast_fp16)[name = string("key_states_19_cast_fp16")]; + tensor x_73_axes_0 = const()[name = string("x_73_axes_0"), val = tensor([1])]; + tensor x_73_cast_fp16 = expand_dims(axes = x_73_axes_0, x = V_layer_cache_9_cast_fp16)[name = string("x_73_cast_fp16")]; + tensor var_3102 = const()[name = string("op_3102"), val = tensor([1, 2, 1, 1])]; + tensor x_75_cast_fp16 = tile(reps = var_3102, x = x_73_cast_fp16)[name = string("x_75_cast_fp16")]; + tensor var_3114 = const()[name = string("op_3114"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_27_cast_fp16 = reshape(shape = var_3114, x = x_75_cast_fp16)[name = string("value_states_27_cast_fp16")]; + bool var_3129_transpose_x_1 = const()[name = string("op_3129_transpose_x_1"), val = bool(false)]; + bool var_3129_transpose_y_1 = const()[name = string("op_3129_transpose_y_1"), val = bool(true)]; + tensor var_3129 = matmul(transpose_x = var_3129_transpose_x_1, transpose_y = var_3129_transpose_y_1, x = query_states_17, y = key_states_19_cast_fp16)[name = string("op_3129")]; + fp16 var_3130_to_fp16 = const()[name = string("op_3130_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_25_cast_fp16 = mul(x = var_3129, y = var_3130_to_fp16)[name = string("attn_weights_25_cast_fp16")]; + tensor attn_weights_27_cast_fp16 = add(x = attn_weights_25_cast_fp16, y = causal_mask)[name = string("attn_weights_27_cast_fp16")]; + int32 var_3165 = const()[name = string("op_3165"), val = int32(-1)]; + tensor attn_weights_29_cast_fp16 = softmax(axis = var_3165, x = attn_weights_27_cast_fp16)[name = string("attn_weights_29_cast_fp16")]; + bool attn_output_41_transpose_x_0 = const()[name = string("attn_output_41_transpose_x_0"), val = bool(false)]; + bool attn_output_41_transpose_y_0 = const()[name = string("attn_output_41_transpose_y_0"), val = bool(false)]; + tensor attn_output_41_cast_fp16 = matmul(transpose_x = attn_output_41_transpose_x_0, transpose_y = attn_output_41_transpose_y_0, x = attn_weights_29_cast_fp16, y = value_states_27_cast_fp16)[name = string("attn_output_41_cast_fp16")]; + tensor var_3176_perm_0 = const()[name = string("op_3176_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_3180 = const()[name = string("op_3180"), val = tensor([1, 1, 2048])]; + tensor var_3176_cast_fp16 = transpose(perm = var_3176_perm_0, x = attn_output_41_cast_fp16)[name = string("transpose_58")]; + tensor attn_output_45_cast_fp16 = reshape(shape = var_3180, x = var_3176_cast_fp16)[name = string("attn_output_45_cast_fp16")]; + tensor var_3185 = const()[name = string("op_3185"), val = tensor([0, 2, 1])]; + string var_3201_pad_type_0 = const()[name = string("op_3201_pad_type_0"), val = string("valid")]; + int32 var_3201_groups_0 = const()[name = string("op_3201_groups_0"), val = int32(1)]; + tensor var_3201_strides_0 = const()[name = string("op_3201_strides_0"), val = tensor([1])]; + tensor var_3201_pad_0 = const()[name = string("op_3201_pad_0"), val = tensor([0, 0])]; + tensor var_3201_dilations_0 = const()[name = string("op_3201_dilations_0"), val = tensor([1])]; + tensor squeeze_4_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680840064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685034432))))[name = string("squeeze_4_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_3186_cast_fp16 = transpose(perm = var_3185, x = attn_output_45_cast_fp16)[name = string("transpose_57")]; + tensor var_3201_cast_fp16 = conv(dilations = var_3201_dilations_0, groups = var_3201_groups_0, pad = var_3201_pad_0, pad_type = var_3201_pad_type_0, strides = var_3201_strides_0, weight = squeeze_4_cast_fp16_to_fp32_to_fp16_palettized, x = var_3186_cast_fp16)[name = string("op_3201_cast_fp16")]; + tensor var_3205 = const()[name = string("op_3205"), val = tensor([0, 2, 1])]; + tensor attn_output_49_cast_fp16 = transpose(perm = var_3205, x = var_3201_cast_fp16)[name = string("transpose_56")]; + tensor hidden_states_49_cast_fp16 = add(x = hidden_states_41_cast_fp16, y = attn_output_49_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; + int32 var_3218 = const()[name = string("op_3218"), val = int32(-1)]; + fp16 const_146_promoted_to_fp16 = const()[name = string("const_146_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3220_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_146_promoted_to_fp16)[name = string("op_3220_cast_fp16")]; + bool input_83_interleave_0 = const()[name = string("input_83_interleave_0"), val = bool(false)]; + tensor input_83_cast_fp16 = concat(axis = var_3218, interleave = input_83_interleave_0, values = (hidden_states_49_cast_fp16, var_3220_cast_fp16))[name = string("input_83_cast_fp16")]; + tensor normed_77_axes_0 = const()[name = string("normed_77_axes_0"), val = tensor([-1])]; + fp16 var_3215_to_fp16 = const()[name = string("op_3215_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_77_cast_fp16 = layer_norm(axes = normed_77_axes_0, epsilon = var_3215_to_fp16, x = input_83_cast_fp16)[name = string("normed_77_cast_fp16")]; + tensor normed_79_begin_0 = const()[name = string("normed_79_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_79_end_0 = const()[name = string("normed_79_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_79_end_mask_0 = const()[name = string("normed_79_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_79_cast_fp16 = slice_by_index(begin = normed_79_begin_0, end = normed_79_end_0, end_mask = normed_79_end_mask_0, x = normed_77_cast_fp16)[name = string("normed_79_cast_fp16")]; + tensor const_149_promoted_to_fp16 = const()[name = string("const_149_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685165568)))]; + tensor x_77_cast_fp16 = mul(x = normed_79_cast_fp16, y = const_149_promoted_to_fp16)[name = string("x_77_cast_fp16")]; + tensor var_3245 = const()[name = string("op_3245"), val = tensor([0, 2, 1])]; + tensor input_85_axes_0 = const()[name = string("input_85_axes_0"), val = tensor([2])]; + tensor var_3246 = transpose(perm = var_3245, x = x_77_cast_fp16)[name = string("transpose_55")]; + tensor input_85 = expand_dims(axes = input_85_axes_0, x = var_3246)[name = string("input_85")]; + string input_87_pad_type_0 = const()[name = string("input_87_pad_type_0"), val = string("valid")]; + tensor input_87_strides_0 = const()[name = string("input_87_strides_0"), val = tensor([1, 1])]; + tensor input_87_pad_0 = const()[name = string("input_87_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_87_dilations_0 = const()[name = string("input_87_dilations_0"), val = tensor([1, 1])]; + int32 input_87_groups_0 = const()[name = string("input_87_groups_0"), val = int32(1)]; + tensor input_87 = conv(dilations = input_87_dilations_0, groups = input_87_groups_0, pad = input_87_pad_0, pad_type = input_87_pad_type_0, strides = input_87_strides_0, weight = model_model_layers_18_mlp_gate_proj_weight_palettized, x = input_85)[name = string("input_87")]; + string b_9_pad_type_0 = const()[name = string("b_9_pad_type_0"), val = string("valid")]; + tensor b_9_strides_0 = const()[name = string("b_9_strides_0"), val = tensor([1, 1])]; + tensor b_9_pad_0 = const()[name = string("b_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_9_dilations_0 = const()[name = string("b_9_dilations_0"), val = tensor([1, 1])]; + int32 b_9_groups_0 = const()[name = string("b_9_groups_0"), val = int32(1)]; + tensor b_9 = conv(dilations = b_9_dilations_0, groups = b_9_groups_0, pad = b_9_pad_0, pad_type = b_9_pad_type_0, strides = b_9_strides_0, weight = model_model_layers_18_mlp_up_proj_weight_palettized, x = input_85)[name = string("b_9")]; + tensor c_9 = silu(x = input_87)[name = string("c_9")]; + tensor input_89 = mul(x = c_9, y = b_9)[name = string("input_89")]; + string e_9_pad_type_0 = const()[name = string("e_9_pad_type_0"), val = string("valid")]; + tensor e_9_strides_0 = const()[name = string("e_9_strides_0"), val = tensor([1, 1])]; + tensor e_9_pad_0 = const()[name = string("e_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_9_dilations_0 = const()[name = string("e_9_dilations_0"), val = tensor([1, 1])]; + int32 e_9_groups_0 = const()[name = string("e_9_groups_0"), val = int32(1)]; + tensor e_9 = conv(dilations = e_9_dilations_0, groups = e_9_groups_0, pad = e_9_pad_0, pad_type = e_9_pad_type_0, strides = e_9_strides_0, weight = model_model_layers_18_mlp_down_proj_weight_palettized, x = input_89)[name = string("e_9")]; + tensor var_3268_axes_0 = const()[name = string("op_3268_axes_0"), val = tensor([2])]; + tensor var_3268 = squeeze(axes = var_3268_axes_0, x = e_9)[name = string("op_3268")]; + tensor var_3269 = const()[name = string("op_3269"), val = tensor([0, 2, 1])]; + tensor var_3270 = transpose(perm = var_3269, x = var_3268)[name = string("transpose_54")]; + tensor hidden_states_51_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = var_3270)[name = string("hidden_states_51_cast_fp16")]; + int32 var_3282 = const()[name = string("op_3282"), val = int32(-1)]; + fp16 const_150_promoted_to_fp16 = const()[name = string("const_150_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3284_cast_fp16 = mul(x = hidden_states_51_cast_fp16, y = const_150_promoted_to_fp16)[name = string("op_3284_cast_fp16")]; + bool input_91_interleave_0 = const()[name = string("input_91_interleave_0"), val = bool(false)]; + tensor input_91_cast_fp16 = concat(axis = var_3282, interleave = input_91_interleave_0, values = (hidden_states_51_cast_fp16, var_3284_cast_fp16))[name = string("input_91_cast_fp16")]; + tensor normed_81_axes_0 = const()[name = string("normed_81_axes_0"), val = tensor([-1])]; + fp16 var_3279_to_fp16 = const()[name = string("op_3279_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_81_cast_fp16 = layer_norm(axes = normed_81_axes_0, epsilon = var_3279_to_fp16, x = input_91_cast_fp16)[name = string("normed_81_cast_fp16")]; + tensor normed_83_begin_0 = const()[name = string("normed_83_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_83_end_0 = const()[name = string("normed_83_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_83_end_mask_0 = const()[name = string("normed_83_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_83_cast_fp16 = slice_by_index(begin = normed_83_begin_0, end = normed_83_end_0, end_mask = normed_83_end_mask_0, x = normed_81_cast_fp16)[name = string("normed_83_cast_fp16")]; + tensor const_153_promoted_to_fp16 = const()[name = string("const_153_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685169728)))]; + tensor hidden_states_53_cast_fp16 = mul(x = normed_83_cast_fp16, y = const_153_promoted_to_fp16)[name = string("hidden_states_53_cast_fp16")]; + tensor var_3301 = const()[name = string("op_3301"), val = tensor([0, 2, 1])]; + tensor var_3304_axes_0 = const()[name = string("op_3304_axes_0"), val = tensor([2])]; + tensor var_3302_cast_fp16 = transpose(perm = var_3301, x = hidden_states_53_cast_fp16)[name = string("transpose_53")]; + tensor var_3304_cast_fp16 = expand_dims(axes = var_3304_axes_0, x = var_3302_cast_fp16)[name = string("op_3304_cast_fp16")]; + string var_3320_pad_type_0 = const()[name = string("op_3320_pad_type_0"), val = string("valid")]; + tensor var_3320_strides_0 = const()[name = string("op_3320_strides_0"), val = tensor([1, 1])]; + tensor var_3320_pad_0 = const()[name = string("op_3320_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3320_dilations_0 = const()[name = string("op_3320_dilations_0"), val = tensor([1, 1])]; + int32 var_3320_groups_0 = const()[name = string("op_3320_groups_0"), val = int32(1)]; + tensor var_3320 = conv(dilations = var_3320_dilations_0, groups = var_3320_groups_0, pad = var_3320_pad_0, pad_type = var_3320_pad_type_0, strides = var_3320_strides_0, weight = model_model_layers_19_self_attn_q_proj_weight_palettized, x = var_3304_cast_fp16)[name = string("op_3320")]; + tensor var_3325 = const()[name = string("op_3325"), val = tensor([1, 16, 1, 128])]; + tensor var_3326 = reshape(shape = var_3325, x = var_3320)[name = string("op_3326")]; + string var_3342_pad_type_0 = const()[name = string("op_3342_pad_type_0"), val = string("valid")]; + tensor var_3342_strides_0 = const()[name = string("op_3342_strides_0"), val = tensor([1, 1])]; + tensor var_3342_pad_0 = const()[name = string("op_3342_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3342_dilations_0 = const()[name = string("op_3342_dilations_0"), val = tensor([1, 1])]; + int32 var_3342_groups_0 = const()[name = string("op_3342_groups_0"), val = int32(1)]; + tensor var_3342 = conv(dilations = var_3342_dilations_0, groups = var_3342_groups_0, pad = var_3342_pad_0, pad_type = var_3342_pad_type_0, strides = var_3342_strides_0, weight = model_model_layers_19_self_attn_k_proj_weight_palettized, x = var_3304_cast_fp16)[name = string("op_3342")]; + tensor var_3347 = const()[name = string("op_3347"), val = tensor([1, 8, 1, 128])]; + tensor var_3348 = reshape(shape = var_3347, x = var_3342)[name = string("op_3348")]; + string var_3364_pad_type_0 = const()[name = string("op_3364_pad_type_0"), val = string("valid")]; + tensor var_3364_strides_0 = const()[name = string("op_3364_strides_0"), val = tensor([1, 1])]; + tensor var_3364_pad_0 = const()[name = string("op_3364_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3364_dilations_0 = const()[name = string("op_3364_dilations_0"), val = tensor([1, 1])]; + int32 var_3364_groups_0 = const()[name = string("op_3364_groups_0"), val = int32(1)]; + tensor var_3364 = conv(dilations = var_3364_dilations_0, groups = var_3364_groups_0, pad = var_3364_pad_0, pad_type = var_3364_pad_type_0, strides = var_3364_strides_0, weight = model_model_layers_19_self_attn_v_proj_weight_palettized, x = var_3304_cast_fp16)[name = string("op_3364")]; + tensor var_3369 = const()[name = string("op_3369"), val = tensor([1, 8, 1, 128])]; + tensor var_3370 = reshape(shape = var_3369, x = var_3364)[name = string("op_3370")]; + int32 var_3385 = const()[name = string("op_3385"), val = int32(-1)]; + fp16 const_154_promoted = const()[name = string("const_154_promoted"), val = fp16(-0x1p+0)]; + tensor var_3387 = mul(x = var_3326, y = const_154_promoted)[name = string("op_3387")]; + bool input_95_interleave_0 = const()[name = string("input_95_interleave_0"), val = bool(false)]; + tensor input_95 = concat(axis = var_3385, interleave = input_95_interleave_0, values = (var_3326, var_3387))[name = string("input_95")]; + tensor normed_85_axes_0 = const()[name = string("normed_85_axes_0"), val = tensor([-1])]; + fp16 var_3382_to_fp16 = const()[name = string("op_3382_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_85_cast_fp16 = layer_norm(axes = normed_85_axes_0, epsilon = var_3382_to_fp16, x = input_95)[name = string("normed_85_cast_fp16")]; + tensor normed_87_begin_0 = const()[name = string("normed_87_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_87_end_0 = const()[name = string("normed_87_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_87_end_mask_0 = const()[name = string("normed_87_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_87 = slice_by_index(begin = normed_87_begin_0, end = normed_87_end_0, end_mask = normed_87_end_mask_0, x = normed_85_cast_fp16)[name = string("normed_87")]; + tensor const_157 = const()[name = string("const_157"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685173888)))]; + tensor q_11 = mul(x = normed_87, y = const_157)[name = string("q_11")]; + int32 var_3410 = const()[name = string("op_3410"), val = int32(-1)]; + fp16 const_158_promoted = const()[name = string("const_158_promoted"), val = fp16(-0x1p+0)]; + tensor var_3412 = mul(x = var_3348, y = const_158_promoted)[name = string("op_3412")]; + bool input_97_interleave_0 = const()[name = string("input_97_interleave_0"), val = bool(false)]; + tensor input_97 = concat(axis = var_3410, interleave = input_97_interleave_0, values = (var_3348, var_3412))[name = string("input_97")]; + tensor normed_89_axes_0 = const()[name = string("normed_89_axes_0"), val = tensor([-1])]; + fp16 var_3407_to_fp16 = const()[name = string("op_3407_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_89_cast_fp16 = layer_norm(axes = normed_89_axes_0, epsilon = var_3407_to_fp16, x = input_97)[name = string("normed_89_cast_fp16")]; + tensor normed_91_begin_0 = const()[name = string("normed_91_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_91_end_0 = const()[name = string("normed_91_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_91_end_mask_0 = const()[name = string("normed_91_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_91 = slice_by_index(begin = normed_91_begin_0, end = normed_91_end_0, end_mask = normed_91_end_mask_0, x = normed_89_cast_fp16)[name = string("normed_91")]; + tensor const_161 = const()[name = string("const_161"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685174208)))]; + tensor k_11 = mul(x = normed_91, y = const_161)[name = string("k_11")]; + tensor var_3426 = mul(x = q_11, y = cos_1_cast_fp16)[name = string("op_3426")]; + tensor x1_21_begin_0 = const()[name = string("x1_21_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_21_end_0 = const()[name = string("x1_21_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_21_end_mask_0 = const()[name = string("x1_21_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_21 = slice_by_index(begin = x1_21_begin_0, end = x1_21_end_0, end_mask = x1_21_end_mask_0, x = q_11)[name = string("x1_21")]; + tensor x2_21_begin_0 = const()[name = string("x2_21_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_21_end_0 = const()[name = string("x2_21_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_21_end_mask_0 = const()[name = string("x2_21_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_21 = slice_by_index(begin = x2_21_begin_0, end = x2_21_end_0, end_mask = x2_21_end_mask_0, x = q_11)[name = string("x2_21")]; + fp16 const_164_promoted = const()[name = string("const_164_promoted"), val = fp16(-0x1p+0)]; + tensor var_3447 = mul(x = x2_21, y = const_164_promoted)[name = string("op_3447")]; + int32 var_3449 = const()[name = string("op_3449"), val = int32(-1)]; + bool var_3450_interleave_0 = const()[name = string("op_3450_interleave_0"), val = bool(false)]; + tensor var_3450 = concat(axis = var_3449, interleave = var_3450_interleave_0, values = (var_3447, x1_21))[name = string("op_3450")]; + tensor var_3451 = mul(x = var_3450, y = sin_1_cast_fp16)[name = string("op_3451")]; + tensor query_states_21 = add(x = var_3426, y = var_3451)[name = string("query_states_21")]; + tensor var_3454 = mul(x = k_11, y = cos_1_cast_fp16)[name = string("op_3454")]; + tensor x1_23_begin_0 = const()[name = string("x1_23_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_23_end_0 = const()[name = string("x1_23_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_23_end_mask_0 = const()[name = string("x1_23_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_23 = slice_by_index(begin = x1_23_begin_0, end = x1_23_end_0, end_mask = x1_23_end_mask_0, x = k_11)[name = string("x1_23")]; + tensor x2_23_begin_0 = const()[name = string("x2_23_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_23_end_0 = const()[name = string("x2_23_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_23_end_mask_0 = const()[name = string("x2_23_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_23 = slice_by_index(begin = x2_23_begin_0, end = x2_23_end_0, end_mask = x2_23_end_mask_0, x = k_11)[name = string("x2_23")]; + fp16 const_167_promoted = const()[name = string("const_167_promoted"), val = fp16(-0x1p+0)]; + tensor var_3475 = mul(x = x2_23, y = const_167_promoted)[name = string("op_3475")]; + int32 var_3477 = const()[name = string("op_3477"), val = int32(-1)]; + bool var_3478_interleave_0 = const()[name = string("op_3478_interleave_0"), val = bool(false)]; + tensor var_3478 = concat(axis = var_3477, interleave = var_3478_interleave_0, values = (var_3475, x1_23))[name = string("op_3478")]; + tensor var_3479 = mul(x = var_3478, y = sin_1_cast_fp16)[name = string("op_3479")]; + tensor key_states_21 = add(x = var_3454, y = var_3479)[name = string("key_states_21")]; + tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([19])]; + tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; + tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; + tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([20])]; + int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; + bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; + tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_60, expand_dims_61, current_pos, expand_dims_63))[name = string("concat_42")]; + tensor concat_43_values1_0 = const()[name = string("concat_43_values1_0"), val = tensor([0])]; + tensor concat_43_values3_0 = const()[name = string("concat_43_values3_0"), val = tensor([0])]; + int32 concat_43_axis_0 = const()[name = string("concat_43_axis_0"), val = int32(0)]; + bool concat_43_interleave_0 = const()[name = string("concat_43_interleave_0"), val = bool(false)]; + tensor concat_43 = concat(axis = concat_43_axis_0, interleave = concat_43_interleave_0, values = (expand_dims_64, concat_43_values1_0, var_1004, concat_43_values3_0))[name = string("concat_43")]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_42, begin_mask = model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0, end = concat_43, end_mask = model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_11_stride_0, update = key_states_21, x = coreml_update_state_37)[name = string("model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_66_write_state")]; + tensor coreml_update_state_38 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_66")]; + tensor expand_dims_66 = const()[name = string("expand_dims_66"), val = tensor([47])]; + tensor expand_dims_67 = const()[name = string("expand_dims_67"), val = tensor([0])]; + tensor expand_dims_69 = const()[name = string("expand_dims_69"), val = tensor([0])]; + tensor expand_dims_70 = const()[name = string("expand_dims_70"), val = tensor([48])]; + int32 concat_46_axis_0 = const()[name = string("concat_46_axis_0"), val = int32(0)]; + bool concat_46_interleave_0 = const()[name = string("concat_46_interleave_0"), val = bool(false)]; + tensor concat_46 = concat(axis = concat_46_axis_0, interleave = concat_46_interleave_0, values = (expand_dims_66, expand_dims_67, current_pos, expand_dims_69))[name = string("concat_46")]; + tensor concat_47_values1_0 = const()[name = string("concat_47_values1_0"), val = tensor([0])]; + tensor concat_47_values3_0 = const()[name = string("concat_47_values3_0"), val = tensor([0])]; + int32 concat_47_axis_0 = const()[name = string("concat_47_axis_0"), val = int32(0)]; + bool concat_47_interleave_0 = const()[name = string("concat_47_interleave_0"), val = bool(false)]; + tensor concat_47 = concat(axis = concat_47_axis_0, interleave = concat_47_interleave_0, values = (expand_dims_70, concat_47_values1_0, var_1004, concat_47_values3_0))[name = string("concat_47")]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_46, begin_mask = model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0, end = concat_47, end_mask = model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_12_stride_0, update = var_3370, x = coreml_update_state_38)[name = string("model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_67_write_state")]; + tensor coreml_update_state_39 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_67")]; + tensor var_3534_begin_0 = const()[name = string("op_3534_begin_0"), val = tensor([19, 0, 0, 0])]; + tensor var_3534_end_0 = const()[name = string("op_3534_end_0"), val = tensor([20, 8, 1024, 128])]; + tensor var_3534_end_mask_0 = const()[name = string("op_3534_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3534_cast_fp16 = slice_by_index(begin = var_3534_begin_0, end = var_3534_end_0, end_mask = var_3534_end_mask_0, x = coreml_update_state_39)[name = string("op_3534_cast_fp16")]; + tensor K_layer_cache_11_axes_0 = const()[name = string("K_layer_cache_11_axes_0"), val = tensor([0])]; + tensor K_layer_cache_11_cast_fp16 = squeeze(axes = K_layer_cache_11_axes_0, x = var_3534_cast_fp16)[name = string("K_layer_cache_11_cast_fp16")]; + tensor var_3541_begin_0 = const()[name = string("op_3541_begin_0"), val = tensor([47, 0, 0, 0])]; + tensor var_3541_end_0 = const()[name = string("op_3541_end_0"), val = tensor([48, 8, 1024, 128])]; + tensor var_3541_end_mask_0 = const()[name = string("op_3541_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3541_cast_fp16 = slice_by_index(begin = var_3541_begin_0, end = var_3541_end_0, end_mask = var_3541_end_mask_0, x = coreml_update_state_39)[name = string("op_3541_cast_fp16")]; + tensor V_layer_cache_11_axes_0 = const()[name = string("V_layer_cache_11_axes_0"), val = tensor([0])]; + tensor V_layer_cache_11_cast_fp16 = squeeze(axes = V_layer_cache_11_axes_0, x = var_3541_cast_fp16)[name = string("V_layer_cache_11_cast_fp16")]; + tensor x_83_axes_0 = const()[name = string("x_83_axes_0"), val = tensor([1])]; + tensor x_83_cast_fp16 = expand_dims(axes = x_83_axes_0, x = K_layer_cache_11_cast_fp16)[name = string("x_83_cast_fp16")]; + tensor var_3578 = const()[name = string("op_3578"), val = tensor([1, 2, 1, 1])]; + tensor x_85_cast_fp16 = tile(reps = var_3578, x = x_83_cast_fp16)[name = string("x_85_cast_fp16")]; + tensor var_3590 = const()[name = string("op_3590"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_23_cast_fp16 = reshape(shape = var_3590, x = x_85_cast_fp16)[name = string("key_states_23_cast_fp16")]; + tensor x_89_axes_0 = const()[name = string("x_89_axes_0"), val = tensor([1])]; + tensor x_89_cast_fp16 = expand_dims(axes = x_89_axes_0, x = V_layer_cache_11_cast_fp16)[name = string("x_89_cast_fp16")]; + tensor var_3598 = const()[name = string("op_3598"), val = tensor([1, 2, 1, 1])]; + tensor x_91_cast_fp16 = tile(reps = var_3598, x = x_89_cast_fp16)[name = string("x_91_cast_fp16")]; + tensor var_3610 = const()[name = string("op_3610"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_33_cast_fp16 = reshape(shape = var_3610, x = x_91_cast_fp16)[name = string("value_states_33_cast_fp16")]; + bool var_3625_transpose_x_1 = const()[name = string("op_3625_transpose_x_1"), val = bool(false)]; + bool var_3625_transpose_y_1 = const()[name = string("op_3625_transpose_y_1"), val = bool(true)]; + tensor var_3625 = matmul(transpose_x = var_3625_transpose_x_1, transpose_y = var_3625_transpose_y_1, x = query_states_21, y = key_states_23_cast_fp16)[name = string("op_3625")]; + fp16 var_3626_to_fp16 = const()[name = string("op_3626_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_31_cast_fp16 = mul(x = var_3625, y = var_3626_to_fp16)[name = string("attn_weights_31_cast_fp16")]; + tensor attn_weights_33_cast_fp16 = add(x = attn_weights_31_cast_fp16, y = causal_mask)[name = string("attn_weights_33_cast_fp16")]; + int32 var_3661 = const()[name = string("op_3661"), val = int32(-1)]; + tensor attn_weights_35_cast_fp16 = softmax(axis = var_3661, x = attn_weights_33_cast_fp16)[name = string("attn_weights_35_cast_fp16")]; + bool attn_output_51_transpose_x_0 = const()[name = string("attn_output_51_transpose_x_0"), val = bool(false)]; + bool attn_output_51_transpose_y_0 = const()[name = string("attn_output_51_transpose_y_0"), val = bool(false)]; + tensor attn_output_51_cast_fp16 = matmul(transpose_x = attn_output_51_transpose_x_0, transpose_y = attn_output_51_transpose_y_0, x = attn_weights_35_cast_fp16, y = value_states_33_cast_fp16)[name = string("attn_output_51_cast_fp16")]; + tensor var_3672_perm_0 = const()[name = string("op_3672_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_3676 = const()[name = string("op_3676"), val = tensor([1, 1, 2048])]; + tensor var_3672_cast_fp16 = transpose(perm = var_3672_perm_0, x = attn_output_51_cast_fp16)[name = string("transpose_52")]; + tensor attn_output_55_cast_fp16 = reshape(shape = var_3676, x = var_3672_cast_fp16)[name = string("attn_output_55_cast_fp16")]; + tensor var_3681 = const()[name = string("op_3681"), val = tensor([0, 2, 1])]; + string var_3697_pad_type_0 = const()[name = string("op_3697_pad_type_0"), val = string("valid")]; + int32 var_3697_groups_0 = const()[name = string("op_3697_groups_0"), val = int32(1)]; + tensor var_3697_strides_0 = const()[name = string("op_3697_strides_0"), val = tensor([1])]; + tensor var_3697_pad_0 = const()[name = string("op_3697_pad_0"), val = tensor([0, 0])]; + tensor var_3697_dilations_0 = const()[name = string("op_3697_dilations_0"), val = tensor([1])]; + tensor squeeze_5_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685174528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689368896))))[name = string("squeeze_5_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_3682_cast_fp16 = transpose(perm = var_3681, x = attn_output_55_cast_fp16)[name = string("transpose_51")]; + tensor var_3697_cast_fp16 = conv(dilations = var_3697_dilations_0, groups = var_3697_groups_0, pad = var_3697_pad_0, pad_type = var_3697_pad_type_0, strides = var_3697_strides_0, weight = squeeze_5_cast_fp16_to_fp32_to_fp16_palettized, x = var_3682_cast_fp16)[name = string("op_3697_cast_fp16")]; + tensor var_3701 = const()[name = string("op_3701"), val = tensor([0, 2, 1])]; + tensor attn_output_59_cast_fp16 = transpose(perm = var_3701, x = var_3697_cast_fp16)[name = string("transpose_50")]; + tensor hidden_states_59_cast_fp16 = add(x = hidden_states_51_cast_fp16, y = attn_output_59_cast_fp16)[name = string("hidden_states_59_cast_fp16")]; + int32 var_3714 = const()[name = string("op_3714"), val = int32(-1)]; + fp16 const_176_promoted_to_fp16 = const()[name = string("const_176_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3716_cast_fp16 = mul(x = hidden_states_59_cast_fp16, y = const_176_promoted_to_fp16)[name = string("op_3716_cast_fp16")]; + bool input_101_interleave_0 = const()[name = string("input_101_interleave_0"), val = bool(false)]; + tensor input_101_cast_fp16 = concat(axis = var_3714, interleave = input_101_interleave_0, values = (hidden_states_59_cast_fp16, var_3716_cast_fp16))[name = string("input_101_cast_fp16")]; + tensor normed_93_axes_0 = const()[name = string("normed_93_axes_0"), val = tensor([-1])]; + fp16 var_3711_to_fp16 = const()[name = string("op_3711_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_93_cast_fp16 = layer_norm(axes = normed_93_axes_0, epsilon = var_3711_to_fp16, x = input_101_cast_fp16)[name = string("normed_93_cast_fp16")]; + tensor normed_95_begin_0 = const()[name = string("normed_95_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_95_end_0 = const()[name = string("normed_95_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_95_end_mask_0 = const()[name = string("normed_95_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_95_cast_fp16 = slice_by_index(begin = normed_95_begin_0, end = normed_95_end_0, end_mask = normed_95_end_mask_0, x = normed_93_cast_fp16)[name = string("normed_95_cast_fp16")]; + tensor const_179_promoted_to_fp16 = const()[name = string("const_179_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689500032)))]; + tensor x_93_cast_fp16 = mul(x = normed_95_cast_fp16, y = const_179_promoted_to_fp16)[name = string("x_93_cast_fp16")]; + tensor var_3741 = const()[name = string("op_3741"), val = tensor([0, 2, 1])]; + tensor input_103_axes_0 = const()[name = string("input_103_axes_0"), val = tensor([2])]; + tensor var_3742 = transpose(perm = var_3741, x = x_93_cast_fp16)[name = string("transpose_49")]; + tensor input_103 = expand_dims(axes = input_103_axes_0, x = var_3742)[name = string("input_103")]; + string input_105_pad_type_0 = const()[name = string("input_105_pad_type_0"), val = string("valid")]; + tensor input_105_strides_0 = const()[name = string("input_105_strides_0"), val = tensor([1, 1])]; + tensor input_105_pad_0 = const()[name = string("input_105_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_105_dilations_0 = const()[name = string("input_105_dilations_0"), val = tensor([1, 1])]; + int32 input_105_groups_0 = const()[name = string("input_105_groups_0"), val = int32(1)]; + tensor input_105 = conv(dilations = input_105_dilations_0, groups = input_105_groups_0, pad = input_105_pad_0, pad_type = input_105_pad_type_0, strides = input_105_strides_0, weight = model_model_layers_19_mlp_gate_proj_weight_palettized, x = input_103)[name = string("input_105")]; + string b_11_pad_type_0 = const()[name = string("b_11_pad_type_0"), val = string("valid")]; + tensor b_11_strides_0 = const()[name = string("b_11_strides_0"), val = tensor([1, 1])]; + tensor b_11_pad_0 = const()[name = string("b_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_11_dilations_0 = const()[name = string("b_11_dilations_0"), val = tensor([1, 1])]; + int32 b_11_groups_0 = const()[name = string("b_11_groups_0"), val = int32(1)]; + tensor b_11 = conv(dilations = b_11_dilations_0, groups = b_11_groups_0, pad = b_11_pad_0, pad_type = b_11_pad_type_0, strides = b_11_strides_0, weight = model_model_layers_19_mlp_up_proj_weight_palettized, x = input_103)[name = string("b_11")]; + tensor c_11 = silu(x = input_105)[name = string("c_11")]; + tensor input_107 = mul(x = c_11, y = b_11)[name = string("input_107")]; + string e_11_pad_type_0 = const()[name = string("e_11_pad_type_0"), val = string("valid")]; + tensor e_11_strides_0 = const()[name = string("e_11_strides_0"), val = tensor([1, 1])]; + tensor e_11_pad_0 = const()[name = string("e_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_11_dilations_0 = const()[name = string("e_11_dilations_0"), val = tensor([1, 1])]; + int32 e_11_groups_0 = const()[name = string("e_11_groups_0"), val = int32(1)]; + tensor e_11 = conv(dilations = e_11_dilations_0, groups = e_11_groups_0, pad = e_11_pad_0, pad_type = e_11_pad_type_0, strides = e_11_strides_0, weight = model_model_layers_19_mlp_down_proj_weight_palettized, x = input_107)[name = string("e_11")]; + tensor var_3764_axes_0 = const()[name = string("op_3764_axes_0"), val = tensor([2])]; + tensor var_3764 = squeeze(axes = var_3764_axes_0, x = e_11)[name = string("op_3764")]; + tensor var_3765 = const()[name = string("op_3765"), val = tensor([0, 2, 1])]; + tensor var_3766 = transpose(perm = var_3765, x = var_3764)[name = string("transpose_48")]; + tensor hidden_states_61_cast_fp16 = add(x = hidden_states_59_cast_fp16, y = var_3766)[name = string("hidden_states_61_cast_fp16")]; + int32 var_3778 = const()[name = string("op_3778"), val = int32(-1)]; + fp16 const_180_promoted_to_fp16 = const()[name = string("const_180_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3780_cast_fp16 = mul(x = hidden_states_61_cast_fp16, y = const_180_promoted_to_fp16)[name = string("op_3780_cast_fp16")]; + bool input_109_interleave_0 = const()[name = string("input_109_interleave_0"), val = bool(false)]; + tensor input_109_cast_fp16 = concat(axis = var_3778, interleave = input_109_interleave_0, values = (hidden_states_61_cast_fp16, var_3780_cast_fp16))[name = string("input_109_cast_fp16")]; + tensor normed_97_axes_0 = const()[name = string("normed_97_axes_0"), val = tensor([-1])]; + fp16 var_3775_to_fp16 = const()[name = string("op_3775_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_97_cast_fp16 = layer_norm(axes = normed_97_axes_0, epsilon = var_3775_to_fp16, x = input_109_cast_fp16)[name = string("normed_97_cast_fp16")]; + tensor normed_99_begin_0 = const()[name = string("normed_99_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_99_end_0 = const()[name = string("normed_99_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_99_end_mask_0 = const()[name = string("normed_99_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_99_cast_fp16 = slice_by_index(begin = normed_99_begin_0, end = normed_99_end_0, end_mask = normed_99_end_mask_0, x = normed_97_cast_fp16)[name = string("normed_99_cast_fp16")]; + tensor const_183_promoted_to_fp16 = const()[name = string("const_183_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689504192)))]; + tensor hidden_states_63_cast_fp16 = mul(x = normed_99_cast_fp16, y = const_183_promoted_to_fp16)[name = string("hidden_states_63_cast_fp16")]; + tensor var_3797 = const()[name = string("op_3797"), val = tensor([0, 2, 1])]; + tensor var_3800_axes_0 = const()[name = string("op_3800_axes_0"), val = tensor([2])]; + tensor var_3798_cast_fp16 = transpose(perm = var_3797, x = hidden_states_63_cast_fp16)[name = string("transpose_47")]; + tensor var_3800_cast_fp16 = expand_dims(axes = var_3800_axes_0, x = var_3798_cast_fp16)[name = string("op_3800_cast_fp16")]; + string var_3816_pad_type_0 = const()[name = string("op_3816_pad_type_0"), val = string("valid")]; + tensor var_3816_strides_0 = const()[name = string("op_3816_strides_0"), val = tensor([1, 1])]; + tensor var_3816_pad_0 = const()[name = string("op_3816_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3816_dilations_0 = const()[name = string("op_3816_dilations_0"), val = tensor([1, 1])]; + int32 var_3816_groups_0 = const()[name = string("op_3816_groups_0"), val = int32(1)]; + tensor var_3816 = conv(dilations = var_3816_dilations_0, groups = var_3816_groups_0, pad = var_3816_pad_0, pad_type = var_3816_pad_type_0, strides = var_3816_strides_0, weight = model_model_layers_20_self_attn_q_proj_weight_palettized, x = var_3800_cast_fp16)[name = string("op_3816")]; + tensor var_3821 = const()[name = string("op_3821"), val = tensor([1, 16, 1, 128])]; + tensor var_3822 = reshape(shape = var_3821, x = var_3816)[name = string("op_3822")]; + string var_3838_pad_type_0 = const()[name = string("op_3838_pad_type_0"), val = string("valid")]; + tensor var_3838_strides_0 = const()[name = string("op_3838_strides_0"), val = tensor([1, 1])]; + tensor var_3838_pad_0 = const()[name = string("op_3838_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3838_dilations_0 = const()[name = string("op_3838_dilations_0"), val = tensor([1, 1])]; + int32 var_3838_groups_0 = const()[name = string("op_3838_groups_0"), val = int32(1)]; + tensor var_3838 = conv(dilations = var_3838_dilations_0, groups = var_3838_groups_0, pad = var_3838_pad_0, pad_type = var_3838_pad_type_0, strides = var_3838_strides_0, weight = model_model_layers_20_self_attn_k_proj_weight_palettized, x = var_3800_cast_fp16)[name = string("op_3838")]; + tensor var_3843 = const()[name = string("op_3843"), val = tensor([1, 8, 1, 128])]; + tensor var_3844 = reshape(shape = var_3843, x = var_3838)[name = string("op_3844")]; + string var_3860_pad_type_0 = const()[name = string("op_3860_pad_type_0"), val = string("valid")]; + tensor var_3860_strides_0 = const()[name = string("op_3860_strides_0"), val = tensor([1, 1])]; + tensor var_3860_pad_0 = const()[name = string("op_3860_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_3860_dilations_0 = const()[name = string("op_3860_dilations_0"), val = tensor([1, 1])]; + int32 var_3860_groups_0 = const()[name = string("op_3860_groups_0"), val = int32(1)]; + tensor var_3860 = conv(dilations = var_3860_dilations_0, groups = var_3860_groups_0, pad = var_3860_pad_0, pad_type = var_3860_pad_type_0, strides = var_3860_strides_0, weight = model_model_layers_20_self_attn_v_proj_weight_palettized, x = var_3800_cast_fp16)[name = string("op_3860")]; + tensor var_3865 = const()[name = string("op_3865"), val = tensor([1, 8, 1, 128])]; + tensor var_3866 = reshape(shape = var_3865, x = var_3860)[name = string("op_3866")]; + int32 var_3881 = const()[name = string("op_3881"), val = int32(-1)]; + fp16 const_184_promoted = const()[name = string("const_184_promoted"), val = fp16(-0x1p+0)]; + tensor var_3883 = mul(x = var_3822, y = const_184_promoted)[name = string("op_3883")]; + bool input_113_interleave_0 = const()[name = string("input_113_interleave_0"), val = bool(false)]; + tensor input_113 = concat(axis = var_3881, interleave = input_113_interleave_0, values = (var_3822, var_3883))[name = string("input_113")]; + tensor normed_101_axes_0 = const()[name = string("normed_101_axes_0"), val = tensor([-1])]; + fp16 var_3878_to_fp16 = const()[name = string("op_3878_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_101_cast_fp16 = layer_norm(axes = normed_101_axes_0, epsilon = var_3878_to_fp16, x = input_113)[name = string("normed_101_cast_fp16")]; + tensor normed_103_begin_0 = const()[name = string("normed_103_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_103_end_0 = const()[name = string("normed_103_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_103_end_mask_0 = const()[name = string("normed_103_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_103 = slice_by_index(begin = normed_103_begin_0, end = normed_103_end_0, end_mask = normed_103_end_mask_0, x = normed_101_cast_fp16)[name = string("normed_103")]; + tensor const_187 = const()[name = string("const_187"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689508352)))]; + tensor q_13 = mul(x = normed_103, y = const_187)[name = string("q_13")]; + int32 var_3906 = const()[name = string("op_3906"), val = int32(-1)]; + fp16 const_188_promoted = const()[name = string("const_188_promoted"), val = fp16(-0x1p+0)]; + tensor var_3908 = mul(x = var_3844, y = const_188_promoted)[name = string("op_3908")]; + bool input_115_interleave_0 = const()[name = string("input_115_interleave_0"), val = bool(false)]; + tensor input_115 = concat(axis = var_3906, interleave = input_115_interleave_0, values = (var_3844, var_3908))[name = string("input_115")]; + tensor normed_105_axes_0 = const()[name = string("normed_105_axes_0"), val = tensor([-1])]; + fp16 var_3903_to_fp16 = const()[name = string("op_3903_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_105_cast_fp16 = layer_norm(axes = normed_105_axes_0, epsilon = var_3903_to_fp16, x = input_115)[name = string("normed_105_cast_fp16")]; + tensor normed_107_begin_0 = const()[name = string("normed_107_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_107_end_0 = const()[name = string("normed_107_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_107_end_mask_0 = const()[name = string("normed_107_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_107 = slice_by_index(begin = normed_107_begin_0, end = normed_107_end_0, end_mask = normed_107_end_mask_0, x = normed_105_cast_fp16)[name = string("normed_107")]; + tensor const_191 = const()[name = string("const_191"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689508672)))]; + tensor k_13 = mul(x = normed_107, y = const_191)[name = string("k_13")]; + tensor var_3922 = mul(x = q_13, y = cos_1_cast_fp16)[name = string("op_3922")]; + tensor x1_25_begin_0 = const()[name = string("x1_25_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_25_end_0 = const()[name = string("x1_25_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_25_end_mask_0 = const()[name = string("x1_25_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_25 = slice_by_index(begin = x1_25_begin_0, end = x1_25_end_0, end_mask = x1_25_end_mask_0, x = q_13)[name = string("x1_25")]; + tensor x2_25_begin_0 = const()[name = string("x2_25_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_25_end_0 = const()[name = string("x2_25_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_25_end_mask_0 = const()[name = string("x2_25_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_25 = slice_by_index(begin = x2_25_begin_0, end = x2_25_end_0, end_mask = x2_25_end_mask_0, x = q_13)[name = string("x2_25")]; + fp16 const_194_promoted = const()[name = string("const_194_promoted"), val = fp16(-0x1p+0)]; + tensor var_3943 = mul(x = x2_25, y = const_194_promoted)[name = string("op_3943")]; + int32 var_3945 = const()[name = string("op_3945"), val = int32(-1)]; + bool var_3946_interleave_0 = const()[name = string("op_3946_interleave_0"), val = bool(false)]; + tensor var_3946 = concat(axis = var_3945, interleave = var_3946_interleave_0, values = (var_3943, x1_25))[name = string("op_3946")]; + tensor var_3947 = mul(x = var_3946, y = sin_1_cast_fp16)[name = string("op_3947")]; + tensor query_states_25 = add(x = var_3922, y = var_3947)[name = string("query_states_25")]; + tensor var_3950 = mul(x = k_13, y = cos_1_cast_fp16)[name = string("op_3950")]; + tensor x1_27_begin_0 = const()[name = string("x1_27_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_27_end_0 = const()[name = string("x1_27_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_27_end_mask_0 = const()[name = string("x1_27_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_27 = slice_by_index(begin = x1_27_begin_0, end = x1_27_end_0, end_mask = x1_27_end_mask_0, x = k_13)[name = string("x1_27")]; + tensor x2_27_begin_0 = const()[name = string("x2_27_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_27_end_0 = const()[name = string("x2_27_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_27_end_mask_0 = const()[name = string("x2_27_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_27 = slice_by_index(begin = x2_27_begin_0, end = x2_27_end_0, end_mask = x2_27_end_mask_0, x = k_13)[name = string("x2_27")]; + fp16 const_197_promoted = const()[name = string("const_197_promoted"), val = fp16(-0x1p+0)]; + tensor var_3971 = mul(x = x2_27, y = const_197_promoted)[name = string("op_3971")]; + int32 var_3973 = const()[name = string("op_3973"), val = int32(-1)]; + bool var_3974_interleave_0 = const()[name = string("op_3974_interleave_0"), val = bool(false)]; + tensor var_3974 = concat(axis = var_3973, interleave = var_3974_interleave_0, values = (var_3971, x1_27))[name = string("op_3974")]; + tensor var_3975 = mul(x = var_3974, y = sin_1_cast_fp16)[name = string("op_3975")]; + tensor key_states_25 = add(x = var_3950, y = var_3975)[name = string("key_states_25")]; + tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([20])]; + tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; + tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; + tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([21])]; + int32 concat_50_axis_0 = const()[name = string("concat_50_axis_0"), val = int32(0)]; + bool concat_50_interleave_0 = const()[name = string("concat_50_interleave_0"), val = bool(false)]; + tensor concat_50 = concat(axis = concat_50_axis_0, interleave = concat_50_interleave_0, values = (expand_dims_72, expand_dims_73, current_pos, expand_dims_75))[name = string("concat_50")]; + tensor concat_51_values1_0 = const()[name = string("concat_51_values1_0"), val = tensor([0])]; + tensor concat_51_values3_0 = const()[name = string("concat_51_values3_0"), val = tensor([0])]; + int32 concat_51_axis_0 = const()[name = string("concat_51_axis_0"), val = int32(0)]; + bool concat_51_interleave_0 = const()[name = string("concat_51_interleave_0"), val = bool(false)]; + tensor concat_51 = concat(axis = concat_51_axis_0, interleave = concat_51_interleave_0, values = (expand_dims_76, concat_51_values1_0, var_1004, concat_51_values3_0))[name = string("concat_51")]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_50, begin_mask = model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0, end = concat_51, end_mask = model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_13_stride_0, update = key_states_25, x = coreml_update_state_39)[name = string("model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_68_write_state")]; + tensor coreml_update_state_40 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_68")]; + tensor expand_dims_78 = const()[name = string("expand_dims_78"), val = tensor([48])]; + tensor expand_dims_79 = const()[name = string("expand_dims_79"), val = tensor([0])]; + tensor expand_dims_81 = const()[name = string("expand_dims_81"), val = tensor([0])]; + tensor expand_dims_82 = const()[name = string("expand_dims_82"), val = tensor([49])]; + int32 concat_54_axis_0 = const()[name = string("concat_54_axis_0"), val = int32(0)]; + bool concat_54_interleave_0 = const()[name = string("concat_54_interleave_0"), val = bool(false)]; + tensor concat_54 = concat(axis = concat_54_axis_0, interleave = concat_54_interleave_0, values = (expand_dims_78, expand_dims_79, current_pos, expand_dims_81))[name = string("concat_54")]; + tensor concat_55_values1_0 = const()[name = string("concat_55_values1_0"), val = tensor([0])]; + tensor concat_55_values3_0 = const()[name = string("concat_55_values3_0"), val = tensor([0])]; + int32 concat_55_axis_0 = const()[name = string("concat_55_axis_0"), val = int32(0)]; + bool concat_55_interleave_0 = const()[name = string("concat_55_interleave_0"), val = bool(false)]; + tensor concat_55 = concat(axis = concat_55_axis_0, interleave = concat_55_interleave_0, values = (expand_dims_82, concat_55_values1_0, var_1004, concat_55_values3_0))[name = string("concat_55")]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_54, begin_mask = model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0, end = concat_55, end_mask = model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_14_stride_0, update = var_3866, x = coreml_update_state_40)[name = string("model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_69_write_state")]; + tensor coreml_update_state_41 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_69")]; + tensor var_4030_begin_0 = const()[name = string("op_4030_begin_0"), val = tensor([20, 0, 0, 0])]; + tensor var_4030_end_0 = const()[name = string("op_4030_end_0"), val = tensor([21, 8, 1024, 128])]; + tensor var_4030_end_mask_0 = const()[name = string("op_4030_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4030_cast_fp16 = slice_by_index(begin = var_4030_begin_0, end = var_4030_end_0, end_mask = var_4030_end_mask_0, x = coreml_update_state_41)[name = string("op_4030_cast_fp16")]; + tensor K_layer_cache_13_axes_0 = const()[name = string("K_layer_cache_13_axes_0"), val = tensor([0])]; + tensor K_layer_cache_13_cast_fp16 = squeeze(axes = K_layer_cache_13_axes_0, x = var_4030_cast_fp16)[name = string("K_layer_cache_13_cast_fp16")]; + tensor var_4037_begin_0 = const()[name = string("op_4037_begin_0"), val = tensor([48, 0, 0, 0])]; + tensor var_4037_end_0 = const()[name = string("op_4037_end_0"), val = tensor([49, 8, 1024, 128])]; + tensor var_4037_end_mask_0 = const()[name = string("op_4037_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4037_cast_fp16 = slice_by_index(begin = var_4037_begin_0, end = var_4037_end_0, end_mask = var_4037_end_mask_0, x = coreml_update_state_41)[name = string("op_4037_cast_fp16")]; + tensor V_layer_cache_13_axes_0 = const()[name = string("V_layer_cache_13_axes_0"), val = tensor([0])]; + tensor V_layer_cache_13_cast_fp16 = squeeze(axes = V_layer_cache_13_axes_0, x = var_4037_cast_fp16)[name = string("V_layer_cache_13_cast_fp16")]; + tensor x_99_axes_0 = const()[name = string("x_99_axes_0"), val = tensor([1])]; + tensor x_99_cast_fp16 = expand_dims(axes = x_99_axes_0, x = K_layer_cache_13_cast_fp16)[name = string("x_99_cast_fp16")]; + tensor var_4074 = const()[name = string("op_4074"), val = tensor([1, 2, 1, 1])]; + tensor x_101_cast_fp16 = tile(reps = var_4074, x = x_99_cast_fp16)[name = string("x_101_cast_fp16")]; + tensor var_4086 = const()[name = string("op_4086"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_27_cast_fp16 = reshape(shape = var_4086, x = x_101_cast_fp16)[name = string("key_states_27_cast_fp16")]; + tensor x_105_axes_0 = const()[name = string("x_105_axes_0"), val = tensor([1])]; + tensor x_105_cast_fp16 = expand_dims(axes = x_105_axes_0, x = V_layer_cache_13_cast_fp16)[name = string("x_105_cast_fp16")]; + tensor var_4094 = const()[name = string("op_4094"), val = tensor([1, 2, 1, 1])]; + tensor x_107_cast_fp16 = tile(reps = var_4094, x = x_105_cast_fp16)[name = string("x_107_cast_fp16")]; + tensor var_4106 = const()[name = string("op_4106"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_39_cast_fp16 = reshape(shape = var_4106, x = x_107_cast_fp16)[name = string("value_states_39_cast_fp16")]; + bool var_4121_transpose_x_1 = const()[name = string("op_4121_transpose_x_1"), val = bool(false)]; + bool var_4121_transpose_y_1 = const()[name = string("op_4121_transpose_y_1"), val = bool(true)]; + tensor var_4121 = matmul(transpose_x = var_4121_transpose_x_1, transpose_y = var_4121_transpose_y_1, x = query_states_25, y = key_states_27_cast_fp16)[name = string("op_4121")]; + fp16 var_4122_to_fp16 = const()[name = string("op_4122_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_37_cast_fp16 = mul(x = var_4121, y = var_4122_to_fp16)[name = string("attn_weights_37_cast_fp16")]; + tensor attn_weights_39_cast_fp16 = add(x = attn_weights_37_cast_fp16, y = causal_mask)[name = string("attn_weights_39_cast_fp16")]; + int32 var_4157 = const()[name = string("op_4157"), val = int32(-1)]; + tensor attn_weights_41_cast_fp16 = softmax(axis = var_4157, x = attn_weights_39_cast_fp16)[name = string("attn_weights_41_cast_fp16")]; + bool attn_output_61_transpose_x_0 = const()[name = string("attn_output_61_transpose_x_0"), val = bool(false)]; + bool attn_output_61_transpose_y_0 = const()[name = string("attn_output_61_transpose_y_0"), val = bool(false)]; + tensor attn_output_61_cast_fp16 = matmul(transpose_x = attn_output_61_transpose_x_0, transpose_y = attn_output_61_transpose_y_0, x = attn_weights_41_cast_fp16, y = value_states_39_cast_fp16)[name = string("attn_output_61_cast_fp16")]; + tensor var_4168_perm_0 = const()[name = string("op_4168_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_4172 = const()[name = string("op_4172"), val = tensor([1, 1, 2048])]; + tensor var_4168_cast_fp16 = transpose(perm = var_4168_perm_0, x = attn_output_61_cast_fp16)[name = string("transpose_46")]; + tensor attn_output_65_cast_fp16 = reshape(shape = var_4172, x = var_4168_cast_fp16)[name = string("attn_output_65_cast_fp16")]; + tensor var_4177 = const()[name = string("op_4177"), val = tensor([0, 2, 1])]; + string var_4193_pad_type_0 = const()[name = string("op_4193_pad_type_0"), val = string("valid")]; + int32 var_4193_groups_0 = const()[name = string("op_4193_groups_0"), val = int32(1)]; + tensor var_4193_strides_0 = const()[name = string("op_4193_strides_0"), val = tensor([1])]; + tensor var_4193_pad_0 = const()[name = string("op_4193_pad_0"), val = tensor([0, 0])]; + tensor var_4193_dilations_0 = const()[name = string("op_4193_dilations_0"), val = tensor([1])]; + tensor squeeze_6_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689508992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693703360))))[name = string("squeeze_6_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_4178_cast_fp16 = transpose(perm = var_4177, x = attn_output_65_cast_fp16)[name = string("transpose_45")]; + tensor var_4193_cast_fp16 = conv(dilations = var_4193_dilations_0, groups = var_4193_groups_0, pad = var_4193_pad_0, pad_type = var_4193_pad_type_0, strides = var_4193_strides_0, weight = squeeze_6_cast_fp16_to_fp32_to_fp16_palettized, x = var_4178_cast_fp16)[name = string("op_4193_cast_fp16")]; + tensor var_4197 = const()[name = string("op_4197"), val = tensor([0, 2, 1])]; + tensor attn_output_69_cast_fp16 = transpose(perm = var_4197, x = var_4193_cast_fp16)[name = string("transpose_44")]; + tensor hidden_states_69_cast_fp16 = add(x = hidden_states_61_cast_fp16, y = attn_output_69_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; + int32 var_4210 = const()[name = string("op_4210"), val = int32(-1)]; + fp16 const_206_promoted_to_fp16 = const()[name = string("const_206_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4212_cast_fp16 = mul(x = hidden_states_69_cast_fp16, y = const_206_promoted_to_fp16)[name = string("op_4212_cast_fp16")]; + bool input_119_interleave_0 = const()[name = string("input_119_interleave_0"), val = bool(false)]; + tensor input_119_cast_fp16 = concat(axis = var_4210, interleave = input_119_interleave_0, values = (hidden_states_69_cast_fp16, var_4212_cast_fp16))[name = string("input_119_cast_fp16")]; + tensor normed_109_axes_0 = const()[name = string("normed_109_axes_0"), val = tensor([-1])]; + fp16 var_4207_to_fp16 = const()[name = string("op_4207_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_109_cast_fp16 = layer_norm(axes = normed_109_axes_0, epsilon = var_4207_to_fp16, x = input_119_cast_fp16)[name = string("normed_109_cast_fp16")]; + tensor normed_111_begin_0 = const()[name = string("normed_111_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_111_end_0 = const()[name = string("normed_111_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_111_end_mask_0 = const()[name = string("normed_111_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_111_cast_fp16 = slice_by_index(begin = normed_111_begin_0, end = normed_111_end_0, end_mask = normed_111_end_mask_0, x = normed_109_cast_fp16)[name = string("normed_111_cast_fp16")]; + tensor const_209_promoted_to_fp16 = const()[name = string("const_209_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693834496)))]; + tensor x_109_cast_fp16 = mul(x = normed_111_cast_fp16, y = const_209_promoted_to_fp16)[name = string("x_109_cast_fp16")]; + tensor var_4237 = const()[name = string("op_4237"), val = tensor([0, 2, 1])]; + tensor input_121_axes_0 = const()[name = string("input_121_axes_0"), val = tensor([2])]; + tensor var_4238 = transpose(perm = var_4237, x = x_109_cast_fp16)[name = string("transpose_43")]; + tensor input_121 = expand_dims(axes = input_121_axes_0, x = var_4238)[name = string("input_121")]; + string input_123_pad_type_0 = const()[name = string("input_123_pad_type_0"), val = string("valid")]; + tensor input_123_strides_0 = const()[name = string("input_123_strides_0"), val = tensor([1, 1])]; + tensor input_123_pad_0 = const()[name = string("input_123_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_123_dilations_0 = const()[name = string("input_123_dilations_0"), val = tensor([1, 1])]; + int32 input_123_groups_0 = const()[name = string("input_123_groups_0"), val = int32(1)]; + tensor input_123 = conv(dilations = input_123_dilations_0, groups = input_123_groups_0, pad = input_123_pad_0, pad_type = input_123_pad_type_0, strides = input_123_strides_0, weight = model_model_layers_20_mlp_gate_proj_weight_palettized, x = input_121)[name = string("input_123")]; + string b_13_pad_type_0 = const()[name = string("b_13_pad_type_0"), val = string("valid")]; + tensor b_13_strides_0 = const()[name = string("b_13_strides_0"), val = tensor([1, 1])]; + tensor b_13_pad_0 = const()[name = string("b_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_13_dilations_0 = const()[name = string("b_13_dilations_0"), val = tensor([1, 1])]; + int32 b_13_groups_0 = const()[name = string("b_13_groups_0"), val = int32(1)]; + tensor b_13 = conv(dilations = b_13_dilations_0, groups = b_13_groups_0, pad = b_13_pad_0, pad_type = b_13_pad_type_0, strides = b_13_strides_0, weight = model_model_layers_20_mlp_up_proj_weight_palettized, x = input_121)[name = string("b_13")]; + tensor c_13 = silu(x = input_123)[name = string("c_13")]; + tensor input_125 = mul(x = c_13, y = b_13)[name = string("input_125")]; + string e_13_pad_type_0 = const()[name = string("e_13_pad_type_0"), val = string("valid")]; + tensor e_13_strides_0 = const()[name = string("e_13_strides_0"), val = tensor([1, 1])]; + tensor e_13_pad_0 = const()[name = string("e_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_13_dilations_0 = const()[name = string("e_13_dilations_0"), val = tensor([1, 1])]; + int32 e_13_groups_0 = const()[name = string("e_13_groups_0"), val = int32(1)]; + tensor e_13 = conv(dilations = e_13_dilations_0, groups = e_13_groups_0, pad = e_13_pad_0, pad_type = e_13_pad_type_0, strides = e_13_strides_0, weight = model_model_layers_20_mlp_down_proj_weight_palettized, x = input_125)[name = string("e_13")]; + tensor var_4260_axes_0 = const()[name = string("op_4260_axes_0"), val = tensor([2])]; + tensor var_4260 = squeeze(axes = var_4260_axes_0, x = e_13)[name = string("op_4260")]; + tensor var_4261 = const()[name = string("op_4261"), val = tensor([0, 2, 1])]; + tensor var_4262 = transpose(perm = var_4261, x = var_4260)[name = string("transpose_42")]; + tensor hidden_states_71_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = var_4262)[name = string("hidden_states_71_cast_fp16")]; + int32 var_4274 = const()[name = string("op_4274"), val = int32(-1)]; + fp16 const_210_promoted_to_fp16 = const()[name = string("const_210_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4276_cast_fp16 = mul(x = hidden_states_71_cast_fp16, y = const_210_promoted_to_fp16)[name = string("op_4276_cast_fp16")]; + bool input_127_interleave_0 = const()[name = string("input_127_interleave_0"), val = bool(false)]; + tensor input_127_cast_fp16 = concat(axis = var_4274, interleave = input_127_interleave_0, values = (hidden_states_71_cast_fp16, var_4276_cast_fp16))[name = string("input_127_cast_fp16")]; + tensor normed_113_axes_0 = const()[name = string("normed_113_axes_0"), val = tensor([-1])]; + fp16 var_4271_to_fp16 = const()[name = string("op_4271_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_113_cast_fp16 = layer_norm(axes = normed_113_axes_0, epsilon = var_4271_to_fp16, x = input_127_cast_fp16)[name = string("normed_113_cast_fp16")]; + tensor normed_115_begin_0 = const()[name = string("normed_115_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_115_end_0 = const()[name = string("normed_115_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_115_end_mask_0 = const()[name = string("normed_115_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_115_cast_fp16 = slice_by_index(begin = normed_115_begin_0, end = normed_115_end_0, end_mask = normed_115_end_mask_0, x = normed_113_cast_fp16)[name = string("normed_115_cast_fp16")]; + tensor const_213_promoted_to_fp16 = const()[name = string("const_213_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693838656)))]; + tensor hidden_states_73_cast_fp16 = mul(x = normed_115_cast_fp16, y = const_213_promoted_to_fp16)[name = string("hidden_states_73_cast_fp16")]; + tensor var_4293 = const()[name = string("op_4293"), val = tensor([0, 2, 1])]; + tensor var_4296_axes_0 = const()[name = string("op_4296_axes_0"), val = tensor([2])]; + tensor var_4294_cast_fp16 = transpose(perm = var_4293, x = hidden_states_73_cast_fp16)[name = string("transpose_41")]; + tensor var_4296_cast_fp16 = expand_dims(axes = var_4296_axes_0, x = var_4294_cast_fp16)[name = string("op_4296_cast_fp16")]; + string var_4312_pad_type_0 = const()[name = string("op_4312_pad_type_0"), val = string("valid")]; + tensor var_4312_strides_0 = const()[name = string("op_4312_strides_0"), val = tensor([1, 1])]; + tensor var_4312_pad_0 = const()[name = string("op_4312_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4312_dilations_0 = const()[name = string("op_4312_dilations_0"), val = tensor([1, 1])]; + int32 var_4312_groups_0 = const()[name = string("op_4312_groups_0"), val = int32(1)]; + tensor var_4312 = conv(dilations = var_4312_dilations_0, groups = var_4312_groups_0, pad = var_4312_pad_0, pad_type = var_4312_pad_type_0, strides = var_4312_strides_0, weight = model_model_layers_21_self_attn_q_proj_weight_palettized, x = var_4296_cast_fp16)[name = string("op_4312")]; + tensor var_4317 = const()[name = string("op_4317"), val = tensor([1, 16, 1, 128])]; + tensor var_4318 = reshape(shape = var_4317, x = var_4312)[name = string("op_4318")]; + string var_4334_pad_type_0 = const()[name = string("op_4334_pad_type_0"), val = string("valid")]; + tensor var_4334_strides_0 = const()[name = string("op_4334_strides_0"), val = tensor([1, 1])]; + tensor var_4334_pad_0 = const()[name = string("op_4334_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4334_dilations_0 = const()[name = string("op_4334_dilations_0"), val = tensor([1, 1])]; + int32 var_4334_groups_0 = const()[name = string("op_4334_groups_0"), val = int32(1)]; + tensor var_4334 = conv(dilations = var_4334_dilations_0, groups = var_4334_groups_0, pad = var_4334_pad_0, pad_type = var_4334_pad_type_0, strides = var_4334_strides_0, weight = model_model_layers_21_self_attn_k_proj_weight_palettized, x = var_4296_cast_fp16)[name = string("op_4334")]; + tensor var_4339 = const()[name = string("op_4339"), val = tensor([1, 8, 1, 128])]; + tensor var_4340 = reshape(shape = var_4339, x = var_4334)[name = string("op_4340")]; + string var_4356_pad_type_0 = const()[name = string("op_4356_pad_type_0"), val = string("valid")]; + tensor var_4356_strides_0 = const()[name = string("op_4356_strides_0"), val = tensor([1, 1])]; + tensor var_4356_pad_0 = const()[name = string("op_4356_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4356_dilations_0 = const()[name = string("op_4356_dilations_0"), val = tensor([1, 1])]; + int32 var_4356_groups_0 = const()[name = string("op_4356_groups_0"), val = int32(1)]; + tensor var_4356 = conv(dilations = var_4356_dilations_0, groups = var_4356_groups_0, pad = var_4356_pad_0, pad_type = var_4356_pad_type_0, strides = var_4356_strides_0, weight = model_model_layers_21_self_attn_v_proj_weight_palettized, x = var_4296_cast_fp16)[name = string("op_4356")]; + tensor var_4361 = const()[name = string("op_4361"), val = tensor([1, 8, 1, 128])]; + tensor var_4362 = reshape(shape = var_4361, x = var_4356)[name = string("op_4362")]; + int32 var_4377 = const()[name = string("op_4377"), val = int32(-1)]; + fp16 const_214_promoted = const()[name = string("const_214_promoted"), val = fp16(-0x1p+0)]; + tensor var_4379 = mul(x = var_4318, y = const_214_promoted)[name = string("op_4379")]; + bool input_131_interleave_0 = const()[name = string("input_131_interleave_0"), val = bool(false)]; + tensor input_131 = concat(axis = var_4377, interleave = input_131_interleave_0, values = (var_4318, var_4379))[name = string("input_131")]; + tensor normed_117_axes_0 = const()[name = string("normed_117_axes_0"), val = tensor([-1])]; + fp16 var_4374_to_fp16 = const()[name = string("op_4374_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_117_cast_fp16 = layer_norm(axes = normed_117_axes_0, epsilon = var_4374_to_fp16, x = input_131)[name = string("normed_117_cast_fp16")]; + tensor normed_119_begin_0 = const()[name = string("normed_119_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_119_end_0 = const()[name = string("normed_119_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_119_end_mask_0 = const()[name = string("normed_119_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_119 = slice_by_index(begin = normed_119_begin_0, end = normed_119_end_0, end_mask = normed_119_end_mask_0, x = normed_117_cast_fp16)[name = string("normed_119")]; + tensor const_217 = const()[name = string("const_217"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693842816)))]; + tensor q_15 = mul(x = normed_119, y = const_217)[name = string("q_15")]; + int32 var_4402 = const()[name = string("op_4402"), val = int32(-1)]; + fp16 const_218_promoted = const()[name = string("const_218_promoted"), val = fp16(-0x1p+0)]; + tensor var_4404 = mul(x = var_4340, y = const_218_promoted)[name = string("op_4404")]; + bool input_133_interleave_0 = const()[name = string("input_133_interleave_0"), val = bool(false)]; + tensor input_133 = concat(axis = var_4402, interleave = input_133_interleave_0, values = (var_4340, var_4404))[name = string("input_133")]; + tensor normed_121_axes_0 = const()[name = string("normed_121_axes_0"), val = tensor([-1])]; + fp16 var_4399_to_fp16 = const()[name = string("op_4399_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_121_cast_fp16 = layer_norm(axes = normed_121_axes_0, epsilon = var_4399_to_fp16, x = input_133)[name = string("normed_121_cast_fp16")]; + tensor normed_123_begin_0 = const()[name = string("normed_123_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_123_end_0 = const()[name = string("normed_123_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_123_end_mask_0 = const()[name = string("normed_123_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_123 = slice_by_index(begin = normed_123_begin_0, end = normed_123_end_0, end_mask = normed_123_end_mask_0, x = normed_121_cast_fp16)[name = string("normed_123")]; + tensor const_221 = const()[name = string("const_221"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693843136)))]; + tensor k_15 = mul(x = normed_123, y = const_221)[name = string("k_15")]; + tensor var_4418 = mul(x = q_15, y = cos_1_cast_fp16)[name = string("op_4418")]; + tensor x1_29_begin_0 = const()[name = string("x1_29_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_29_end_0 = const()[name = string("x1_29_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_29_end_mask_0 = const()[name = string("x1_29_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_29 = slice_by_index(begin = x1_29_begin_0, end = x1_29_end_0, end_mask = x1_29_end_mask_0, x = q_15)[name = string("x1_29")]; + tensor x2_29_begin_0 = const()[name = string("x2_29_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_29_end_0 = const()[name = string("x2_29_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_29_end_mask_0 = const()[name = string("x2_29_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_29 = slice_by_index(begin = x2_29_begin_0, end = x2_29_end_0, end_mask = x2_29_end_mask_0, x = q_15)[name = string("x2_29")]; + fp16 const_224_promoted = const()[name = string("const_224_promoted"), val = fp16(-0x1p+0)]; + tensor var_4439 = mul(x = x2_29, y = const_224_promoted)[name = string("op_4439")]; + int32 var_4441 = const()[name = string("op_4441"), val = int32(-1)]; + bool var_4442_interleave_0 = const()[name = string("op_4442_interleave_0"), val = bool(false)]; + tensor var_4442 = concat(axis = var_4441, interleave = var_4442_interleave_0, values = (var_4439, x1_29))[name = string("op_4442")]; + tensor var_4443 = mul(x = var_4442, y = sin_1_cast_fp16)[name = string("op_4443")]; + tensor query_states_29 = add(x = var_4418, y = var_4443)[name = string("query_states_29")]; + tensor var_4446 = mul(x = k_15, y = cos_1_cast_fp16)[name = string("op_4446")]; + tensor x1_31_begin_0 = const()[name = string("x1_31_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_31_end_0 = const()[name = string("x1_31_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_31_end_mask_0 = const()[name = string("x1_31_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_31 = slice_by_index(begin = x1_31_begin_0, end = x1_31_end_0, end_mask = x1_31_end_mask_0, x = k_15)[name = string("x1_31")]; + tensor x2_31_begin_0 = const()[name = string("x2_31_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_31_end_0 = const()[name = string("x2_31_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_31_end_mask_0 = const()[name = string("x2_31_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_31 = slice_by_index(begin = x2_31_begin_0, end = x2_31_end_0, end_mask = x2_31_end_mask_0, x = k_15)[name = string("x2_31")]; + fp16 const_227_promoted = const()[name = string("const_227_promoted"), val = fp16(-0x1p+0)]; + tensor var_4467 = mul(x = x2_31, y = const_227_promoted)[name = string("op_4467")]; + int32 var_4469 = const()[name = string("op_4469"), val = int32(-1)]; + bool var_4470_interleave_0 = const()[name = string("op_4470_interleave_0"), val = bool(false)]; + tensor var_4470 = concat(axis = var_4469, interleave = var_4470_interleave_0, values = (var_4467, x1_31))[name = string("op_4470")]; + tensor var_4471 = mul(x = var_4470, y = sin_1_cast_fp16)[name = string("op_4471")]; + tensor key_states_29 = add(x = var_4446, y = var_4471)[name = string("key_states_29")]; + tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([21])]; + tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; + tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; + tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([22])]; + int32 concat_58_axis_0 = const()[name = string("concat_58_axis_0"), val = int32(0)]; + bool concat_58_interleave_0 = const()[name = string("concat_58_interleave_0"), val = bool(false)]; + tensor concat_58 = concat(axis = concat_58_axis_0, interleave = concat_58_interleave_0, values = (expand_dims_84, expand_dims_85, current_pos, expand_dims_87))[name = string("concat_58")]; + tensor concat_59_values1_0 = const()[name = string("concat_59_values1_0"), val = tensor([0])]; + tensor concat_59_values3_0 = const()[name = string("concat_59_values3_0"), val = tensor([0])]; + int32 concat_59_axis_0 = const()[name = string("concat_59_axis_0"), val = int32(0)]; + bool concat_59_interleave_0 = const()[name = string("concat_59_interleave_0"), val = bool(false)]; + tensor concat_59 = concat(axis = concat_59_axis_0, interleave = concat_59_interleave_0, values = (expand_dims_88, concat_59_values1_0, var_1004, concat_59_values3_0))[name = string("concat_59")]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_58, begin_mask = model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0, end = concat_59, end_mask = model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_15_stride_0, update = key_states_29, x = coreml_update_state_41)[name = string("model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_70_write_state")]; + tensor coreml_update_state_42 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_70")]; + tensor expand_dims_90 = const()[name = string("expand_dims_90"), val = tensor([49])]; + tensor expand_dims_91 = const()[name = string("expand_dims_91"), val = tensor([0])]; + tensor expand_dims_93 = const()[name = string("expand_dims_93"), val = tensor([0])]; + tensor expand_dims_94 = const()[name = string("expand_dims_94"), val = tensor([50])]; + int32 concat_62_axis_0 = const()[name = string("concat_62_axis_0"), val = int32(0)]; + bool concat_62_interleave_0 = const()[name = string("concat_62_interleave_0"), val = bool(false)]; + tensor concat_62 = concat(axis = concat_62_axis_0, interleave = concat_62_interleave_0, values = (expand_dims_90, expand_dims_91, current_pos, expand_dims_93))[name = string("concat_62")]; + tensor concat_63_values1_0 = const()[name = string("concat_63_values1_0"), val = tensor([0])]; + tensor concat_63_values3_0 = const()[name = string("concat_63_values3_0"), val = tensor([0])]; + int32 concat_63_axis_0 = const()[name = string("concat_63_axis_0"), val = int32(0)]; + bool concat_63_interleave_0 = const()[name = string("concat_63_interleave_0"), val = bool(false)]; + tensor concat_63 = concat(axis = concat_63_axis_0, interleave = concat_63_interleave_0, values = (expand_dims_94, concat_63_values1_0, var_1004, concat_63_values3_0))[name = string("concat_63")]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_62, begin_mask = model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0, end = concat_63, end_mask = model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_16_stride_0, update = var_4362, x = coreml_update_state_42)[name = string("model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_71_write_state")]; + tensor coreml_update_state_43 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_71")]; + tensor var_4526_begin_0 = const()[name = string("op_4526_begin_0"), val = tensor([21, 0, 0, 0])]; + tensor var_4526_end_0 = const()[name = string("op_4526_end_0"), val = tensor([22, 8, 1024, 128])]; + tensor var_4526_end_mask_0 = const()[name = string("op_4526_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4526_cast_fp16 = slice_by_index(begin = var_4526_begin_0, end = var_4526_end_0, end_mask = var_4526_end_mask_0, x = coreml_update_state_43)[name = string("op_4526_cast_fp16")]; + tensor K_layer_cache_15_axes_0 = const()[name = string("K_layer_cache_15_axes_0"), val = tensor([0])]; + tensor K_layer_cache_15_cast_fp16 = squeeze(axes = K_layer_cache_15_axes_0, x = var_4526_cast_fp16)[name = string("K_layer_cache_15_cast_fp16")]; + tensor var_4533_begin_0 = const()[name = string("op_4533_begin_0"), val = tensor([49, 0, 0, 0])]; + tensor var_4533_end_0 = const()[name = string("op_4533_end_0"), val = tensor([50, 8, 1024, 128])]; + tensor var_4533_end_mask_0 = const()[name = string("op_4533_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4533_cast_fp16 = slice_by_index(begin = var_4533_begin_0, end = var_4533_end_0, end_mask = var_4533_end_mask_0, x = coreml_update_state_43)[name = string("op_4533_cast_fp16")]; + tensor V_layer_cache_15_axes_0 = const()[name = string("V_layer_cache_15_axes_0"), val = tensor([0])]; + tensor V_layer_cache_15_cast_fp16 = squeeze(axes = V_layer_cache_15_axes_0, x = var_4533_cast_fp16)[name = string("V_layer_cache_15_cast_fp16")]; + tensor x_115_axes_0 = const()[name = string("x_115_axes_0"), val = tensor([1])]; + tensor x_115_cast_fp16 = expand_dims(axes = x_115_axes_0, x = K_layer_cache_15_cast_fp16)[name = string("x_115_cast_fp16")]; + tensor var_4570 = const()[name = string("op_4570"), val = tensor([1, 2, 1, 1])]; + tensor x_117_cast_fp16 = tile(reps = var_4570, x = x_115_cast_fp16)[name = string("x_117_cast_fp16")]; + tensor var_4582 = const()[name = string("op_4582"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_31_cast_fp16 = reshape(shape = var_4582, x = x_117_cast_fp16)[name = string("key_states_31_cast_fp16")]; + tensor x_121_axes_0 = const()[name = string("x_121_axes_0"), val = tensor([1])]; + tensor x_121_cast_fp16 = expand_dims(axes = x_121_axes_0, x = V_layer_cache_15_cast_fp16)[name = string("x_121_cast_fp16")]; + tensor var_4590 = const()[name = string("op_4590"), val = tensor([1, 2, 1, 1])]; + tensor x_123_cast_fp16 = tile(reps = var_4590, x = x_121_cast_fp16)[name = string("x_123_cast_fp16")]; + tensor var_4602 = const()[name = string("op_4602"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_45_cast_fp16 = reshape(shape = var_4602, x = x_123_cast_fp16)[name = string("value_states_45_cast_fp16")]; + bool var_4617_transpose_x_1 = const()[name = string("op_4617_transpose_x_1"), val = bool(false)]; + bool var_4617_transpose_y_1 = const()[name = string("op_4617_transpose_y_1"), val = bool(true)]; + tensor var_4617 = matmul(transpose_x = var_4617_transpose_x_1, transpose_y = var_4617_transpose_y_1, x = query_states_29, y = key_states_31_cast_fp16)[name = string("op_4617")]; + fp16 var_4618_to_fp16 = const()[name = string("op_4618_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_43_cast_fp16 = mul(x = var_4617, y = var_4618_to_fp16)[name = string("attn_weights_43_cast_fp16")]; + tensor attn_weights_45_cast_fp16 = add(x = attn_weights_43_cast_fp16, y = causal_mask)[name = string("attn_weights_45_cast_fp16")]; + int32 var_4653 = const()[name = string("op_4653"), val = int32(-1)]; + tensor attn_weights_47_cast_fp16 = softmax(axis = var_4653, x = attn_weights_45_cast_fp16)[name = string("attn_weights_47_cast_fp16")]; + bool attn_output_71_transpose_x_0 = const()[name = string("attn_output_71_transpose_x_0"), val = bool(false)]; + bool attn_output_71_transpose_y_0 = const()[name = string("attn_output_71_transpose_y_0"), val = bool(false)]; + tensor attn_output_71_cast_fp16 = matmul(transpose_x = attn_output_71_transpose_x_0, transpose_y = attn_output_71_transpose_y_0, x = attn_weights_47_cast_fp16, y = value_states_45_cast_fp16)[name = string("attn_output_71_cast_fp16")]; + tensor var_4664_perm_0 = const()[name = string("op_4664_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_4668 = const()[name = string("op_4668"), val = tensor([1, 1, 2048])]; + tensor var_4664_cast_fp16 = transpose(perm = var_4664_perm_0, x = attn_output_71_cast_fp16)[name = string("transpose_40")]; + tensor attn_output_75_cast_fp16 = reshape(shape = var_4668, x = var_4664_cast_fp16)[name = string("attn_output_75_cast_fp16")]; + tensor var_4673 = const()[name = string("op_4673"), val = tensor([0, 2, 1])]; + string var_4689_pad_type_0 = const()[name = string("op_4689_pad_type_0"), val = string("valid")]; + int32 var_4689_groups_0 = const()[name = string("op_4689_groups_0"), val = int32(1)]; + tensor var_4689_strides_0 = const()[name = string("op_4689_strides_0"), val = tensor([1])]; + tensor var_4689_pad_0 = const()[name = string("op_4689_pad_0"), val = tensor([0, 0])]; + tensor var_4689_dilations_0 = const()[name = string("op_4689_dilations_0"), val = tensor([1])]; + tensor squeeze_7_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693843456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698037824))))[name = string("squeeze_7_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_4674_cast_fp16 = transpose(perm = var_4673, x = attn_output_75_cast_fp16)[name = string("transpose_39")]; + tensor var_4689_cast_fp16 = conv(dilations = var_4689_dilations_0, groups = var_4689_groups_0, pad = var_4689_pad_0, pad_type = var_4689_pad_type_0, strides = var_4689_strides_0, weight = squeeze_7_cast_fp16_to_fp32_to_fp16_palettized, x = var_4674_cast_fp16)[name = string("op_4689_cast_fp16")]; + tensor var_4693 = const()[name = string("op_4693"), val = tensor([0, 2, 1])]; + tensor attn_output_79_cast_fp16 = transpose(perm = var_4693, x = var_4689_cast_fp16)[name = string("transpose_38")]; + tensor hidden_states_79_cast_fp16 = add(x = hidden_states_71_cast_fp16, y = attn_output_79_cast_fp16)[name = string("hidden_states_79_cast_fp16")]; + int32 var_4706 = const()[name = string("op_4706"), val = int32(-1)]; + fp16 const_236_promoted_to_fp16 = const()[name = string("const_236_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4708_cast_fp16 = mul(x = hidden_states_79_cast_fp16, y = const_236_promoted_to_fp16)[name = string("op_4708_cast_fp16")]; + bool input_137_interleave_0 = const()[name = string("input_137_interleave_0"), val = bool(false)]; + tensor input_137_cast_fp16 = concat(axis = var_4706, interleave = input_137_interleave_0, values = (hidden_states_79_cast_fp16, var_4708_cast_fp16))[name = string("input_137_cast_fp16")]; + tensor normed_125_axes_0 = const()[name = string("normed_125_axes_0"), val = tensor([-1])]; + fp16 var_4703_to_fp16 = const()[name = string("op_4703_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_125_cast_fp16 = layer_norm(axes = normed_125_axes_0, epsilon = var_4703_to_fp16, x = input_137_cast_fp16)[name = string("normed_125_cast_fp16")]; + tensor normed_127_begin_0 = const()[name = string("normed_127_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_127_end_0 = const()[name = string("normed_127_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_127_end_mask_0 = const()[name = string("normed_127_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_127_cast_fp16 = slice_by_index(begin = normed_127_begin_0, end = normed_127_end_0, end_mask = normed_127_end_mask_0, x = normed_125_cast_fp16)[name = string("normed_127_cast_fp16")]; + tensor const_239_promoted_to_fp16 = const()[name = string("const_239_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698168960)))]; + tensor x_125_cast_fp16 = mul(x = normed_127_cast_fp16, y = const_239_promoted_to_fp16)[name = string("x_125_cast_fp16")]; + tensor var_4733 = const()[name = string("op_4733"), val = tensor([0, 2, 1])]; + tensor input_139_axes_0 = const()[name = string("input_139_axes_0"), val = tensor([2])]; + tensor var_4734 = transpose(perm = var_4733, x = x_125_cast_fp16)[name = string("transpose_37")]; + tensor input_139 = expand_dims(axes = input_139_axes_0, x = var_4734)[name = string("input_139")]; + string input_141_pad_type_0 = const()[name = string("input_141_pad_type_0"), val = string("valid")]; + tensor input_141_strides_0 = const()[name = string("input_141_strides_0"), val = tensor([1, 1])]; + tensor input_141_pad_0 = const()[name = string("input_141_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_141_dilations_0 = const()[name = string("input_141_dilations_0"), val = tensor([1, 1])]; + int32 input_141_groups_0 = const()[name = string("input_141_groups_0"), val = int32(1)]; + tensor input_141 = conv(dilations = input_141_dilations_0, groups = input_141_groups_0, pad = input_141_pad_0, pad_type = input_141_pad_type_0, strides = input_141_strides_0, weight = model_model_layers_21_mlp_gate_proj_weight_palettized, x = input_139)[name = string("input_141")]; + string b_15_pad_type_0 = const()[name = string("b_15_pad_type_0"), val = string("valid")]; + tensor b_15_strides_0 = const()[name = string("b_15_strides_0"), val = tensor([1, 1])]; + tensor b_15_pad_0 = const()[name = string("b_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_15_dilations_0 = const()[name = string("b_15_dilations_0"), val = tensor([1, 1])]; + int32 b_15_groups_0 = const()[name = string("b_15_groups_0"), val = int32(1)]; + tensor b_15 = conv(dilations = b_15_dilations_0, groups = b_15_groups_0, pad = b_15_pad_0, pad_type = b_15_pad_type_0, strides = b_15_strides_0, weight = model_model_layers_21_mlp_up_proj_weight_palettized, x = input_139)[name = string("b_15")]; + tensor c_15 = silu(x = input_141)[name = string("c_15")]; + tensor input_143 = mul(x = c_15, y = b_15)[name = string("input_143")]; + string e_15_pad_type_0 = const()[name = string("e_15_pad_type_0"), val = string("valid")]; + tensor e_15_strides_0 = const()[name = string("e_15_strides_0"), val = tensor([1, 1])]; + tensor e_15_pad_0 = const()[name = string("e_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_15_dilations_0 = const()[name = string("e_15_dilations_0"), val = tensor([1, 1])]; + int32 e_15_groups_0 = const()[name = string("e_15_groups_0"), val = int32(1)]; + tensor e_15 = conv(dilations = e_15_dilations_0, groups = e_15_groups_0, pad = e_15_pad_0, pad_type = e_15_pad_type_0, strides = e_15_strides_0, weight = model_model_layers_21_mlp_down_proj_weight_palettized, x = input_143)[name = string("e_15")]; + tensor var_4756_axes_0 = const()[name = string("op_4756_axes_0"), val = tensor([2])]; + tensor var_4756 = squeeze(axes = var_4756_axes_0, x = e_15)[name = string("op_4756")]; + tensor var_4757 = const()[name = string("op_4757"), val = tensor([0, 2, 1])]; + tensor var_4758 = transpose(perm = var_4757, x = var_4756)[name = string("transpose_36")]; + tensor hidden_states_81_cast_fp16 = add(x = hidden_states_79_cast_fp16, y = var_4758)[name = string("hidden_states_81_cast_fp16")]; + int32 var_4770 = const()[name = string("op_4770"), val = int32(-1)]; + fp16 const_240_promoted_to_fp16 = const()[name = string("const_240_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4772_cast_fp16 = mul(x = hidden_states_81_cast_fp16, y = const_240_promoted_to_fp16)[name = string("op_4772_cast_fp16")]; + bool input_145_interleave_0 = const()[name = string("input_145_interleave_0"), val = bool(false)]; + tensor input_145_cast_fp16 = concat(axis = var_4770, interleave = input_145_interleave_0, values = (hidden_states_81_cast_fp16, var_4772_cast_fp16))[name = string("input_145_cast_fp16")]; + tensor normed_129_axes_0 = const()[name = string("normed_129_axes_0"), val = tensor([-1])]; + fp16 var_4767_to_fp16 = const()[name = string("op_4767_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_129_cast_fp16 = layer_norm(axes = normed_129_axes_0, epsilon = var_4767_to_fp16, x = input_145_cast_fp16)[name = string("normed_129_cast_fp16")]; + tensor normed_131_begin_0 = const()[name = string("normed_131_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_131_end_0 = const()[name = string("normed_131_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_131_end_mask_0 = const()[name = string("normed_131_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_131_cast_fp16 = slice_by_index(begin = normed_131_begin_0, end = normed_131_end_0, end_mask = normed_131_end_mask_0, x = normed_129_cast_fp16)[name = string("normed_131_cast_fp16")]; + tensor const_243_promoted_to_fp16 = const()[name = string("const_243_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698173120)))]; + tensor hidden_states_83_cast_fp16 = mul(x = normed_131_cast_fp16, y = const_243_promoted_to_fp16)[name = string("hidden_states_83_cast_fp16")]; + tensor var_4789 = const()[name = string("op_4789"), val = tensor([0, 2, 1])]; + tensor var_4792_axes_0 = const()[name = string("op_4792_axes_0"), val = tensor([2])]; + tensor var_4790_cast_fp16 = transpose(perm = var_4789, x = hidden_states_83_cast_fp16)[name = string("transpose_35")]; + tensor var_4792_cast_fp16 = expand_dims(axes = var_4792_axes_0, x = var_4790_cast_fp16)[name = string("op_4792_cast_fp16")]; + string var_4808_pad_type_0 = const()[name = string("op_4808_pad_type_0"), val = string("valid")]; + tensor var_4808_strides_0 = const()[name = string("op_4808_strides_0"), val = tensor([1, 1])]; + tensor var_4808_pad_0 = const()[name = string("op_4808_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4808_dilations_0 = const()[name = string("op_4808_dilations_0"), val = tensor([1, 1])]; + int32 var_4808_groups_0 = const()[name = string("op_4808_groups_0"), val = int32(1)]; + tensor var_4808 = conv(dilations = var_4808_dilations_0, groups = var_4808_groups_0, pad = var_4808_pad_0, pad_type = var_4808_pad_type_0, strides = var_4808_strides_0, weight = model_model_layers_22_self_attn_q_proj_weight_palettized, x = var_4792_cast_fp16)[name = string("op_4808")]; + tensor var_4813 = const()[name = string("op_4813"), val = tensor([1, 16, 1, 128])]; + tensor var_4814 = reshape(shape = var_4813, x = var_4808)[name = string("op_4814")]; + string var_4830_pad_type_0 = const()[name = string("op_4830_pad_type_0"), val = string("valid")]; + tensor var_4830_strides_0 = const()[name = string("op_4830_strides_0"), val = tensor([1, 1])]; + tensor var_4830_pad_0 = const()[name = string("op_4830_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4830_dilations_0 = const()[name = string("op_4830_dilations_0"), val = tensor([1, 1])]; + int32 var_4830_groups_0 = const()[name = string("op_4830_groups_0"), val = int32(1)]; + tensor var_4830 = conv(dilations = var_4830_dilations_0, groups = var_4830_groups_0, pad = var_4830_pad_0, pad_type = var_4830_pad_type_0, strides = var_4830_strides_0, weight = model_model_layers_22_self_attn_k_proj_weight_palettized, x = var_4792_cast_fp16)[name = string("op_4830")]; + tensor var_4835 = const()[name = string("op_4835"), val = tensor([1, 8, 1, 128])]; + tensor var_4836 = reshape(shape = var_4835, x = var_4830)[name = string("op_4836")]; + string var_4852_pad_type_0 = const()[name = string("op_4852_pad_type_0"), val = string("valid")]; + tensor var_4852_strides_0 = const()[name = string("op_4852_strides_0"), val = tensor([1, 1])]; + tensor var_4852_pad_0 = const()[name = string("op_4852_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_4852_dilations_0 = const()[name = string("op_4852_dilations_0"), val = tensor([1, 1])]; + int32 var_4852_groups_0 = const()[name = string("op_4852_groups_0"), val = int32(1)]; + tensor var_4852 = conv(dilations = var_4852_dilations_0, groups = var_4852_groups_0, pad = var_4852_pad_0, pad_type = var_4852_pad_type_0, strides = var_4852_strides_0, weight = model_model_layers_22_self_attn_v_proj_weight_palettized, x = var_4792_cast_fp16)[name = string("op_4852")]; + tensor var_4857 = const()[name = string("op_4857"), val = tensor([1, 8, 1, 128])]; + tensor var_4858 = reshape(shape = var_4857, x = var_4852)[name = string("op_4858")]; + int32 var_4873 = const()[name = string("op_4873"), val = int32(-1)]; + fp16 const_244_promoted = const()[name = string("const_244_promoted"), val = fp16(-0x1p+0)]; + tensor var_4875 = mul(x = var_4814, y = const_244_promoted)[name = string("op_4875")]; + bool input_149_interleave_0 = const()[name = string("input_149_interleave_0"), val = bool(false)]; + tensor input_149 = concat(axis = var_4873, interleave = input_149_interleave_0, values = (var_4814, var_4875))[name = string("input_149")]; + tensor normed_133_axes_0 = const()[name = string("normed_133_axes_0"), val = tensor([-1])]; + fp16 var_4870_to_fp16 = const()[name = string("op_4870_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_133_cast_fp16 = layer_norm(axes = normed_133_axes_0, epsilon = var_4870_to_fp16, x = input_149)[name = string("normed_133_cast_fp16")]; + tensor normed_135_begin_0 = const()[name = string("normed_135_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_135_end_0 = const()[name = string("normed_135_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_135_end_mask_0 = const()[name = string("normed_135_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_135 = slice_by_index(begin = normed_135_begin_0, end = normed_135_end_0, end_mask = normed_135_end_mask_0, x = normed_133_cast_fp16)[name = string("normed_135")]; + tensor const_247 = const()[name = string("const_247"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698177280)))]; + tensor q_17 = mul(x = normed_135, y = const_247)[name = string("q_17")]; + int32 var_4898 = const()[name = string("op_4898"), val = int32(-1)]; + fp16 const_248_promoted = const()[name = string("const_248_promoted"), val = fp16(-0x1p+0)]; + tensor var_4900 = mul(x = var_4836, y = const_248_promoted)[name = string("op_4900")]; + bool input_151_interleave_0 = const()[name = string("input_151_interleave_0"), val = bool(false)]; + tensor input_151 = concat(axis = var_4898, interleave = input_151_interleave_0, values = (var_4836, var_4900))[name = string("input_151")]; + tensor normed_137_axes_0 = const()[name = string("normed_137_axes_0"), val = tensor([-1])]; + fp16 var_4895_to_fp16 = const()[name = string("op_4895_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_137_cast_fp16 = layer_norm(axes = normed_137_axes_0, epsilon = var_4895_to_fp16, x = input_151)[name = string("normed_137_cast_fp16")]; + tensor normed_139_begin_0 = const()[name = string("normed_139_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_139_end_0 = const()[name = string("normed_139_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_139_end_mask_0 = const()[name = string("normed_139_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_139 = slice_by_index(begin = normed_139_begin_0, end = normed_139_end_0, end_mask = normed_139_end_mask_0, x = normed_137_cast_fp16)[name = string("normed_139")]; + tensor const_251 = const()[name = string("const_251"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698177600)))]; + tensor k_17 = mul(x = normed_139, y = const_251)[name = string("k_17")]; + tensor var_4914 = mul(x = q_17, y = cos_1_cast_fp16)[name = string("op_4914")]; + tensor x1_33_begin_0 = const()[name = string("x1_33_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_33_end_0 = const()[name = string("x1_33_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_33_end_mask_0 = const()[name = string("x1_33_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_33 = slice_by_index(begin = x1_33_begin_0, end = x1_33_end_0, end_mask = x1_33_end_mask_0, x = q_17)[name = string("x1_33")]; + tensor x2_33_begin_0 = const()[name = string("x2_33_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_33_end_0 = const()[name = string("x2_33_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_33_end_mask_0 = const()[name = string("x2_33_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_33 = slice_by_index(begin = x2_33_begin_0, end = x2_33_end_0, end_mask = x2_33_end_mask_0, x = q_17)[name = string("x2_33")]; + fp16 const_254_promoted = const()[name = string("const_254_promoted"), val = fp16(-0x1p+0)]; + tensor var_4935 = mul(x = x2_33, y = const_254_promoted)[name = string("op_4935")]; + int32 var_4937 = const()[name = string("op_4937"), val = int32(-1)]; + bool var_4938_interleave_0 = const()[name = string("op_4938_interleave_0"), val = bool(false)]; + tensor var_4938 = concat(axis = var_4937, interleave = var_4938_interleave_0, values = (var_4935, x1_33))[name = string("op_4938")]; + tensor var_4939 = mul(x = var_4938, y = sin_1_cast_fp16)[name = string("op_4939")]; + tensor query_states_33 = add(x = var_4914, y = var_4939)[name = string("query_states_33")]; + tensor var_4942 = mul(x = k_17, y = cos_1_cast_fp16)[name = string("op_4942")]; + tensor x1_35_begin_0 = const()[name = string("x1_35_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_35_end_0 = const()[name = string("x1_35_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_35_end_mask_0 = const()[name = string("x1_35_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_35 = slice_by_index(begin = x1_35_begin_0, end = x1_35_end_0, end_mask = x1_35_end_mask_0, x = k_17)[name = string("x1_35")]; + tensor x2_35_begin_0 = const()[name = string("x2_35_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_35_end_0 = const()[name = string("x2_35_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_35_end_mask_0 = const()[name = string("x2_35_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_35 = slice_by_index(begin = x2_35_begin_0, end = x2_35_end_0, end_mask = x2_35_end_mask_0, x = k_17)[name = string("x2_35")]; + fp16 const_257_promoted = const()[name = string("const_257_promoted"), val = fp16(-0x1p+0)]; + tensor var_4963 = mul(x = x2_35, y = const_257_promoted)[name = string("op_4963")]; + int32 var_4965 = const()[name = string("op_4965"), val = int32(-1)]; + bool var_4966_interleave_0 = const()[name = string("op_4966_interleave_0"), val = bool(false)]; + tensor var_4966 = concat(axis = var_4965, interleave = var_4966_interleave_0, values = (var_4963, x1_35))[name = string("op_4966")]; + tensor var_4967 = mul(x = var_4966, y = sin_1_cast_fp16)[name = string("op_4967")]; + tensor key_states_33 = add(x = var_4942, y = var_4967)[name = string("key_states_33")]; + tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([22])]; + tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; + tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; + tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([23])]; + int32 concat_66_axis_0 = const()[name = string("concat_66_axis_0"), val = int32(0)]; + bool concat_66_interleave_0 = const()[name = string("concat_66_interleave_0"), val = bool(false)]; + tensor concat_66 = concat(axis = concat_66_axis_0, interleave = concat_66_interleave_0, values = (expand_dims_96, expand_dims_97, current_pos, expand_dims_99))[name = string("concat_66")]; + tensor concat_67_values1_0 = const()[name = string("concat_67_values1_0"), val = tensor([0])]; + tensor concat_67_values3_0 = const()[name = string("concat_67_values3_0"), val = tensor([0])]; + int32 concat_67_axis_0 = const()[name = string("concat_67_axis_0"), val = int32(0)]; + bool concat_67_interleave_0 = const()[name = string("concat_67_interleave_0"), val = bool(false)]; + tensor concat_67 = concat(axis = concat_67_axis_0, interleave = concat_67_interleave_0, values = (expand_dims_100, concat_67_values1_0, var_1004, concat_67_values3_0))[name = string("concat_67")]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_66, begin_mask = model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0, end = concat_67, end_mask = model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_17_stride_0, update = key_states_33, x = coreml_update_state_43)[name = string("model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_72_write_state")]; + tensor coreml_update_state_44 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_72")]; + tensor expand_dims_102 = const()[name = string("expand_dims_102"), val = tensor([50])]; + tensor expand_dims_103 = const()[name = string("expand_dims_103"), val = tensor([0])]; + tensor expand_dims_105 = const()[name = string("expand_dims_105"), val = tensor([0])]; + tensor expand_dims_106 = const()[name = string("expand_dims_106"), val = tensor([51])]; + int32 concat_70_axis_0 = const()[name = string("concat_70_axis_0"), val = int32(0)]; + bool concat_70_interleave_0 = const()[name = string("concat_70_interleave_0"), val = bool(false)]; + tensor concat_70 = concat(axis = concat_70_axis_0, interleave = concat_70_interleave_0, values = (expand_dims_102, expand_dims_103, current_pos, expand_dims_105))[name = string("concat_70")]; + tensor concat_71_values1_0 = const()[name = string("concat_71_values1_0"), val = tensor([0])]; + tensor concat_71_values3_0 = const()[name = string("concat_71_values3_0"), val = tensor([0])]; + int32 concat_71_axis_0 = const()[name = string("concat_71_axis_0"), val = int32(0)]; + bool concat_71_interleave_0 = const()[name = string("concat_71_interleave_0"), val = bool(false)]; + tensor concat_71 = concat(axis = concat_71_axis_0, interleave = concat_71_interleave_0, values = (expand_dims_106, concat_71_values1_0, var_1004, concat_71_values3_0))[name = string("concat_71")]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_70, begin_mask = model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0, end = concat_71, end_mask = model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_18_stride_0, update = var_4858, x = coreml_update_state_44)[name = string("model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_73_write_state")]; + tensor coreml_update_state_45 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_73")]; + tensor var_5022_begin_0 = const()[name = string("op_5022_begin_0"), val = tensor([22, 0, 0, 0])]; + tensor var_5022_end_0 = const()[name = string("op_5022_end_0"), val = tensor([23, 8, 1024, 128])]; + tensor var_5022_end_mask_0 = const()[name = string("op_5022_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_5022_cast_fp16 = slice_by_index(begin = var_5022_begin_0, end = var_5022_end_0, end_mask = var_5022_end_mask_0, x = coreml_update_state_45)[name = string("op_5022_cast_fp16")]; + tensor K_layer_cache_17_axes_0 = const()[name = string("K_layer_cache_17_axes_0"), val = tensor([0])]; + tensor K_layer_cache_17_cast_fp16 = squeeze(axes = K_layer_cache_17_axes_0, x = var_5022_cast_fp16)[name = string("K_layer_cache_17_cast_fp16")]; + tensor var_5029_begin_0 = const()[name = string("op_5029_begin_0"), val = tensor([50, 0, 0, 0])]; + tensor var_5029_end_0 = const()[name = string("op_5029_end_0"), val = tensor([51, 8, 1024, 128])]; + tensor var_5029_end_mask_0 = const()[name = string("op_5029_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_5029_cast_fp16 = slice_by_index(begin = var_5029_begin_0, end = var_5029_end_0, end_mask = var_5029_end_mask_0, x = coreml_update_state_45)[name = string("op_5029_cast_fp16")]; + tensor V_layer_cache_17_axes_0 = const()[name = string("V_layer_cache_17_axes_0"), val = tensor([0])]; + tensor V_layer_cache_17_cast_fp16 = squeeze(axes = V_layer_cache_17_axes_0, x = var_5029_cast_fp16)[name = string("V_layer_cache_17_cast_fp16")]; + tensor x_131_axes_0 = const()[name = string("x_131_axes_0"), val = tensor([1])]; + tensor x_131_cast_fp16 = expand_dims(axes = x_131_axes_0, x = K_layer_cache_17_cast_fp16)[name = string("x_131_cast_fp16")]; + tensor var_5066 = const()[name = string("op_5066"), val = tensor([1, 2, 1, 1])]; + tensor x_133_cast_fp16 = tile(reps = var_5066, x = x_131_cast_fp16)[name = string("x_133_cast_fp16")]; + tensor var_5078 = const()[name = string("op_5078"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_35_cast_fp16 = reshape(shape = var_5078, x = x_133_cast_fp16)[name = string("key_states_35_cast_fp16")]; + tensor x_137_axes_0 = const()[name = string("x_137_axes_0"), val = tensor([1])]; + tensor x_137_cast_fp16 = expand_dims(axes = x_137_axes_0, x = V_layer_cache_17_cast_fp16)[name = string("x_137_cast_fp16")]; + tensor var_5086 = const()[name = string("op_5086"), val = tensor([1, 2, 1, 1])]; + tensor x_139_cast_fp16 = tile(reps = var_5086, x = x_137_cast_fp16)[name = string("x_139_cast_fp16")]; + tensor var_5098 = const()[name = string("op_5098"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_51_cast_fp16 = reshape(shape = var_5098, x = x_139_cast_fp16)[name = string("value_states_51_cast_fp16")]; + bool var_5113_transpose_x_1 = const()[name = string("op_5113_transpose_x_1"), val = bool(false)]; + bool var_5113_transpose_y_1 = const()[name = string("op_5113_transpose_y_1"), val = bool(true)]; + tensor var_5113 = matmul(transpose_x = var_5113_transpose_x_1, transpose_y = var_5113_transpose_y_1, x = query_states_33, y = key_states_35_cast_fp16)[name = string("op_5113")]; + fp16 var_5114_to_fp16 = const()[name = string("op_5114_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_49_cast_fp16 = mul(x = var_5113, y = var_5114_to_fp16)[name = string("attn_weights_49_cast_fp16")]; + tensor attn_weights_51_cast_fp16 = add(x = attn_weights_49_cast_fp16, y = causal_mask)[name = string("attn_weights_51_cast_fp16")]; + int32 var_5149 = const()[name = string("op_5149"), val = int32(-1)]; + tensor attn_weights_53_cast_fp16 = softmax(axis = var_5149, x = attn_weights_51_cast_fp16)[name = string("attn_weights_53_cast_fp16")]; + bool attn_output_81_transpose_x_0 = const()[name = string("attn_output_81_transpose_x_0"), val = bool(false)]; + bool attn_output_81_transpose_y_0 = const()[name = string("attn_output_81_transpose_y_0"), val = bool(false)]; + tensor attn_output_81_cast_fp16 = matmul(transpose_x = attn_output_81_transpose_x_0, transpose_y = attn_output_81_transpose_y_0, x = attn_weights_53_cast_fp16, y = value_states_51_cast_fp16)[name = string("attn_output_81_cast_fp16")]; + tensor var_5160_perm_0 = const()[name = string("op_5160_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_5164 = const()[name = string("op_5164"), val = tensor([1, 1, 2048])]; + tensor var_5160_cast_fp16 = transpose(perm = var_5160_perm_0, x = attn_output_81_cast_fp16)[name = string("transpose_34")]; + tensor attn_output_85_cast_fp16 = reshape(shape = var_5164, x = var_5160_cast_fp16)[name = string("attn_output_85_cast_fp16")]; + tensor var_5169 = const()[name = string("op_5169"), val = tensor([0, 2, 1])]; + string var_5185_pad_type_0 = const()[name = string("op_5185_pad_type_0"), val = string("valid")]; + int32 var_5185_groups_0 = const()[name = string("op_5185_groups_0"), val = int32(1)]; + tensor var_5185_strides_0 = const()[name = string("op_5185_strides_0"), val = tensor([1])]; + tensor var_5185_pad_0 = const()[name = string("op_5185_pad_0"), val = tensor([0, 0])]; + tensor var_5185_dilations_0 = const()[name = string("op_5185_dilations_0"), val = tensor([1])]; + tensor squeeze_8_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698177920))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702372288))))[name = string("squeeze_8_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_5170_cast_fp16 = transpose(perm = var_5169, x = attn_output_85_cast_fp16)[name = string("transpose_33")]; + tensor var_5185_cast_fp16 = conv(dilations = var_5185_dilations_0, groups = var_5185_groups_0, pad = var_5185_pad_0, pad_type = var_5185_pad_type_0, strides = var_5185_strides_0, weight = squeeze_8_cast_fp16_to_fp32_to_fp16_palettized, x = var_5170_cast_fp16)[name = string("op_5185_cast_fp16")]; + tensor var_5189 = const()[name = string("op_5189"), val = tensor([0, 2, 1])]; + tensor attn_output_89_cast_fp16 = transpose(perm = var_5189, x = var_5185_cast_fp16)[name = string("transpose_32")]; + tensor hidden_states_89_cast_fp16 = add(x = hidden_states_81_cast_fp16, y = attn_output_89_cast_fp16)[name = string("hidden_states_89_cast_fp16")]; + int32 var_5202 = const()[name = string("op_5202"), val = int32(-1)]; + fp16 const_266_promoted_to_fp16 = const()[name = string("const_266_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5204_cast_fp16 = mul(x = hidden_states_89_cast_fp16, y = const_266_promoted_to_fp16)[name = string("op_5204_cast_fp16")]; + bool input_155_interleave_0 = const()[name = string("input_155_interleave_0"), val = bool(false)]; + tensor input_155_cast_fp16 = concat(axis = var_5202, interleave = input_155_interleave_0, values = (hidden_states_89_cast_fp16, var_5204_cast_fp16))[name = string("input_155_cast_fp16")]; + tensor normed_141_axes_0 = const()[name = string("normed_141_axes_0"), val = tensor([-1])]; + fp16 var_5199_to_fp16 = const()[name = string("op_5199_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_141_cast_fp16 = layer_norm(axes = normed_141_axes_0, epsilon = var_5199_to_fp16, x = input_155_cast_fp16)[name = string("normed_141_cast_fp16")]; + tensor normed_143_begin_0 = const()[name = string("normed_143_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_143_end_0 = const()[name = string("normed_143_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_143_end_mask_0 = const()[name = string("normed_143_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_143_cast_fp16 = slice_by_index(begin = normed_143_begin_0, end = normed_143_end_0, end_mask = normed_143_end_mask_0, x = normed_141_cast_fp16)[name = string("normed_143_cast_fp16")]; + tensor const_269_promoted_to_fp16 = const()[name = string("const_269_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702503424)))]; + tensor x_141_cast_fp16 = mul(x = normed_143_cast_fp16, y = const_269_promoted_to_fp16)[name = string("x_141_cast_fp16")]; + tensor var_5229 = const()[name = string("op_5229"), val = tensor([0, 2, 1])]; + tensor input_157_axes_0 = const()[name = string("input_157_axes_0"), val = tensor([2])]; + tensor var_5230 = transpose(perm = var_5229, x = x_141_cast_fp16)[name = string("transpose_31")]; + tensor input_157 = expand_dims(axes = input_157_axes_0, x = var_5230)[name = string("input_157")]; + string input_159_pad_type_0 = const()[name = string("input_159_pad_type_0"), val = string("valid")]; + tensor input_159_strides_0 = const()[name = string("input_159_strides_0"), val = tensor([1, 1])]; + tensor input_159_pad_0 = const()[name = string("input_159_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_159_dilations_0 = const()[name = string("input_159_dilations_0"), val = tensor([1, 1])]; + int32 input_159_groups_0 = const()[name = string("input_159_groups_0"), val = int32(1)]; + tensor input_159 = conv(dilations = input_159_dilations_0, groups = input_159_groups_0, pad = input_159_pad_0, pad_type = input_159_pad_type_0, strides = input_159_strides_0, weight = model_model_layers_22_mlp_gate_proj_weight_palettized, x = input_157)[name = string("input_159")]; + string b_17_pad_type_0 = const()[name = string("b_17_pad_type_0"), val = string("valid")]; + tensor b_17_strides_0 = const()[name = string("b_17_strides_0"), val = tensor([1, 1])]; + tensor b_17_pad_0 = const()[name = string("b_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_17_dilations_0 = const()[name = string("b_17_dilations_0"), val = tensor([1, 1])]; + int32 b_17_groups_0 = const()[name = string("b_17_groups_0"), val = int32(1)]; + tensor b_17 = conv(dilations = b_17_dilations_0, groups = b_17_groups_0, pad = b_17_pad_0, pad_type = b_17_pad_type_0, strides = b_17_strides_0, weight = model_model_layers_22_mlp_up_proj_weight_palettized, x = input_157)[name = string("b_17")]; + tensor c_17 = silu(x = input_159)[name = string("c_17")]; + tensor input_161 = mul(x = c_17, y = b_17)[name = string("input_161")]; + string e_17_pad_type_0 = const()[name = string("e_17_pad_type_0"), val = string("valid")]; + tensor e_17_strides_0 = const()[name = string("e_17_strides_0"), val = tensor([1, 1])]; + tensor e_17_pad_0 = const()[name = string("e_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_17_dilations_0 = const()[name = string("e_17_dilations_0"), val = tensor([1, 1])]; + int32 e_17_groups_0 = const()[name = string("e_17_groups_0"), val = int32(1)]; + tensor e_17 = conv(dilations = e_17_dilations_0, groups = e_17_groups_0, pad = e_17_pad_0, pad_type = e_17_pad_type_0, strides = e_17_strides_0, weight = model_model_layers_22_mlp_down_proj_weight_palettized, x = input_161)[name = string("e_17")]; + tensor var_5252_axes_0 = const()[name = string("op_5252_axes_0"), val = tensor([2])]; + tensor var_5252 = squeeze(axes = var_5252_axes_0, x = e_17)[name = string("op_5252")]; + tensor var_5253 = const()[name = string("op_5253"), val = tensor([0, 2, 1])]; + tensor var_5254 = transpose(perm = var_5253, x = var_5252)[name = string("transpose_30")]; + tensor hidden_states_91_cast_fp16 = add(x = hidden_states_89_cast_fp16, y = var_5254)[name = string("hidden_states_91_cast_fp16")]; + int32 var_5266 = const()[name = string("op_5266"), val = int32(-1)]; + fp16 const_270_promoted_to_fp16 = const()[name = string("const_270_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5268_cast_fp16 = mul(x = hidden_states_91_cast_fp16, y = const_270_promoted_to_fp16)[name = string("op_5268_cast_fp16")]; + bool input_163_interleave_0 = const()[name = string("input_163_interleave_0"), val = bool(false)]; + tensor input_163_cast_fp16 = concat(axis = var_5266, interleave = input_163_interleave_0, values = (hidden_states_91_cast_fp16, var_5268_cast_fp16))[name = string("input_163_cast_fp16")]; + tensor normed_145_axes_0 = const()[name = string("normed_145_axes_0"), val = tensor([-1])]; + fp16 var_5263_to_fp16 = const()[name = string("op_5263_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_145_cast_fp16 = layer_norm(axes = normed_145_axes_0, epsilon = var_5263_to_fp16, x = input_163_cast_fp16)[name = string("normed_145_cast_fp16")]; + tensor normed_147_begin_0 = const()[name = string("normed_147_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_147_end_0 = const()[name = string("normed_147_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_147_end_mask_0 = const()[name = string("normed_147_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_147_cast_fp16 = slice_by_index(begin = normed_147_begin_0, end = normed_147_end_0, end_mask = normed_147_end_mask_0, x = normed_145_cast_fp16)[name = string("normed_147_cast_fp16")]; + tensor const_273_promoted_to_fp16 = const()[name = string("const_273_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702507584)))]; + tensor hidden_states_93_cast_fp16 = mul(x = normed_147_cast_fp16, y = const_273_promoted_to_fp16)[name = string("hidden_states_93_cast_fp16")]; + tensor var_5285 = const()[name = string("op_5285"), val = tensor([0, 2, 1])]; + tensor var_5288_axes_0 = const()[name = string("op_5288_axes_0"), val = tensor([2])]; + tensor var_5286_cast_fp16 = transpose(perm = var_5285, x = hidden_states_93_cast_fp16)[name = string("transpose_29")]; + tensor var_5288_cast_fp16 = expand_dims(axes = var_5288_axes_0, x = var_5286_cast_fp16)[name = string("op_5288_cast_fp16")]; + string var_5304_pad_type_0 = const()[name = string("op_5304_pad_type_0"), val = string("valid")]; + tensor var_5304_strides_0 = const()[name = string("op_5304_strides_0"), val = tensor([1, 1])]; + tensor var_5304_pad_0 = const()[name = string("op_5304_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5304_dilations_0 = const()[name = string("op_5304_dilations_0"), val = tensor([1, 1])]; + int32 var_5304_groups_0 = const()[name = string("op_5304_groups_0"), val = int32(1)]; + tensor var_5304 = conv(dilations = var_5304_dilations_0, groups = var_5304_groups_0, pad = var_5304_pad_0, pad_type = var_5304_pad_type_0, strides = var_5304_strides_0, weight = model_model_layers_23_self_attn_q_proj_weight_palettized, x = var_5288_cast_fp16)[name = string("op_5304")]; + tensor var_5309 = const()[name = string("op_5309"), val = tensor([1, 16, 1, 128])]; + tensor var_5310 = reshape(shape = var_5309, x = var_5304)[name = string("op_5310")]; + string var_5326_pad_type_0 = const()[name = string("op_5326_pad_type_0"), val = string("valid")]; + tensor var_5326_strides_0 = const()[name = string("op_5326_strides_0"), val = tensor([1, 1])]; + tensor var_5326_pad_0 = const()[name = string("op_5326_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5326_dilations_0 = const()[name = string("op_5326_dilations_0"), val = tensor([1, 1])]; + int32 var_5326_groups_0 = const()[name = string("op_5326_groups_0"), val = int32(1)]; + tensor var_5326 = conv(dilations = var_5326_dilations_0, groups = var_5326_groups_0, pad = var_5326_pad_0, pad_type = var_5326_pad_type_0, strides = var_5326_strides_0, weight = model_model_layers_23_self_attn_k_proj_weight_palettized, x = var_5288_cast_fp16)[name = string("op_5326")]; + tensor var_5331 = const()[name = string("op_5331"), val = tensor([1, 8, 1, 128])]; + tensor var_5332 = reshape(shape = var_5331, x = var_5326)[name = string("op_5332")]; + string var_5348_pad_type_0 = const()[name = string("op_5348_pad_type_0"), val = string("valid")]; + tensor var_5348_strides_0 = const()[name = string("op_5348_strides_0"), val = tensor([1, 1])]; + tensor var_5348_pad_0 = const()[name = string("op_5348_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5348_dilations_0 = const()[name = string("op_5348_dilations_0"), val = tensor([1, 1])]; + int32 var_5348_groups_0 = const()[name = string("op_5348_groups_0"), val = int32(1)]; + tensor var_5348 = conv(dilations = var_5348_dilations_0, groups = var_5348_groups_0, pad = var_5348_pad_0, pad_type = var_5348_pad_type_0, strides = var_5348_strides_0, weight = model_model_layers_23_self_attn_v_proj_weight_palettized, x = var_5288_cast_fp16)[name = string("op_5348")]; + tensor var_5353 = const()[name = string("op_5353"), val = tensor([1, 8, 1, 128])]; + tensor var_5354 = reshape(shape = var_5353, x = var_5348)[name = string("op_5354")]; + int32 var_5369 = const()[name = string("op_5369"), val = int32(-1)]; + fp16 const_274_promoted = const()[name = string("const_274_promoted"), val = fp16(-0x1p+0)]; + tensor var_5371 = mul(x = var_5310, y = const_274_promoted)[name = string("op_5371")]; + bool input_167_interleave_0 = const()[name = string("input_167_interleave_0"), val = bool(false)]; + tensor input_167 = concat(axis = var_5369, interleave = input_167_interleave_0, values = (var_5310, var_5371))[name = string("input_167")]; + tensor normed_149_axes_0 = const()[name = string("normed_149_axes_0"), val = tensor([-1])]; + fp16 var_5366_to_fp16 = const()[name = string("op_5366_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_149_cast_fp16 = layer_norm(axes = normed_149_axes_0, epsilon = var_5366_to_fp16, x = input_167)[name = string("normed_149_cast_fp16")]; + tensor normed_151_begin_0 = const()[name = string("normed_151_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_151_end_0 = const()[name = string("normed_151_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_151_end_mask_0 = const()[name = string("normed_151_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_151 = slice_by_index(begin = normed_151_begin_0, end = normed_151_end_0, end_mask = normed_151_end_mask_0, x = normed_149_cast_fp16)[name = string("normed_151")]; + tensor const_277 = const()[name = string("const_277"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702511744)))]; + tensor q_19 = mul(x = normed_151, y = const_277)[name = string("q_19")]; + int32 var_5394 = const()[name = string("op_5394"), val = int32(-1)]; + fp16 const_278_promoted = const()[name = string("const_278_promoted"), val = fp16(-0x1p+0)]; + tensor var_5396 = mul(x = var_5332, y = const_278_promoted)[name = string("op_5396")]; + bool input_169_interleave_0 = const()[name = string("input_169_interleave_0"), val = bool(false)]; + tensor input_169 = concat(axis = var_5394, interleave = input_169_interleave_0, values = (var_5332, var_5396))[name = string("input_169")]; + tensor normed_153_axes_0 = const()[name = string("normed_153_axes_0"), val = tensor([-1])]; + fp16 var_5391_to_fp16 = const()[name = string("op_5391_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_153_cast_fp16 = layer_norm(axes = normed_153_axes_0, epsilon = var_5391_to_fp16, x = input_169)[name = string("normed_153_cast_fp16")]; + tensor normed_155_begin_0 = const()[name = string("normed_155_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_155_end_0 = const()[name = string("normed_155_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_155_end_mask_0 = const()[name = string("normed_155_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_155 = slice_by_index(begin = normed_155_begin_0, end = normed_155_end_0, end_mask = normed_155_end_mask_0, x = normed_153_cast_fp16)[name = string("normed_155")]; + tensor const_281 = const()[name = string("const_281"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702512064)))]; + tensor k_19 = mul(x = normed_155, y = const_281)[name = string("k_19")]; + tensor var_5410 = mul(x = q_19, y = cos_1_cast_fp16)[name = string("op_5410")]; + tensor x1_37_begin_0 = const()[name = string("x1_37_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_37_end_0 = const()[name = string("x1_37_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_37_end_mask_0 = const()[name = string("x1_37_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_37 = slice_by_index(begin = x1_37_begin_0, end = x1_37_end_0, end_mask = x1_37_end_mask_0, x = q_19)[name = string("x1_37")]; + tensor x2_37_begin_0 = const()[name = string("x2_37_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_37_end_0 = const()[name = string("x2_37_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_37_end_mask_0 = const()[name = string("x2_37_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_37 = slice_by_index(begin = x2_37_begin_0, end = x2_37_end_0, end_mask = x2_37_end_mask_0, x = q_19)[name = string("x2_37")]; + fp16 const_284_promoted = const()[name = string("const_284_promoted"), val = fp16(-0x1p+0)]; + tensor var_5431 = mul(x = x2_37, y = const_284_promoted)[name = string("op_5431")]; + int32 var_5433 = const()[name = string("op_5433"), val = int32(-1)]; + bool var_5434_interleave_0 = const()[name = string("op_5434_interleave_0"), val = bool(false)]; + tensor var_5434 = concat(axis = var_5433, interleave = var_5434_interleave_0, values = (var_5431, x1_37))[name = string("op_5434")]; + tensor var_5435 = mul(x = var_5434, y = sin_1_cast_fp16)[name = string("op_5435")]; + tensor query_states_37 = add(x = var_5410, y = var_5435)[name = string("query_states_37")]; + tensor var_5438 = mul(x = k_19, y = cos_1_cast_fp16)[name = string("op_5438")]; + tensor x1_39_begin_0 = const()[name = string("x1_39_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_39_end_0 = const()[name = string("x1_39_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_39_end_mask_0 = const()[name = string("x1_39_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_39 = slice_by_index(begin = x1_39_begin_0, end = x1_39_end_0, end_mask = x1_39_end_mask_0, x = k_19)[name = string("x1_39")]; + tensor x2_39_begin_0 = const()[name = string("x2_39_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_39_end_0 = const()[name = string("x2_39_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_39_end_mask_0 = const()[name = string("x2_39_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_39 = slice_by_index(begin = x2_39_begin_0, end = x2_39_end_0, end_mask = x2_39_end_mask_0, x = k_19)[name = string("x2_39")]; + fp16 const_287_promoted = const()[name = string("const_287_promoted"), val = fp16(-0x1p+0)]; + tensor var_5459 = mul(x = x2_39, y = const_287_promoted)[name = string("op_5459")]; + int32 var_5461 = const()[name = string("op_5461"), val = int32(-1)]; + bool var_5462_interleave_0 = const()[name = string("op_5462_interleave_0"), val = bool(false)]; + tensor var_5462 = concat(axis = var_5461, interleave = var_5462_interleave_0, values = (var_5459, x1_39))[name = string("op_5462")]; + tensor var_5463 = mul(x = var_5462, y = sin_1_cast_fp16)[name = string("op_5463")]; + tensor key_states_37 = add(x = var_5438, y = var_5463)[name = string("key_states_37")]; + tensor expand_dims_108 = const()[name = string("expand_dims_108"), val = tensor([23])]; + tensor expand_dims_109 = const()[name = string("expand_dims_109"), val = tensor([0])]; + tensor expand_dims_111 = const()[name = string("expand_dims_111"), val = tensor([0])]; + tensor expand_dims_112 = const()[name = string("expand_dims_112"), val = tensor([24])]; + int32 concat_74_axis_0 = const()[name = string("concat_74_axis_0"), val = int32(0)]; + bool concat_74_interleave_0 = const()[name = string("concat_74_interleave_0"), val = bool(false)]; + tensor concat_74 = concat(axis = concat_74_axis_0, interleave = concat_74_interleave_0, values = (expand_dims_108, expand_dims_109, current_pos, expand_dims_111))[name = string("concat_74")]; + tensor concat_75_values1_0 = const()[name = string("concat_75_values1_0"), val = tensor([0])]; + tensor concat_75_values3_0 = const()[name = string("concat_75_values3_0"), val = tensor([0])]; + int32 concat_75_axis_0 = const()[name = string("concat_75_axis_0"), val = int32(0)]; + bool concat_75_interleave_0 = const()[name = string("concat_75_interleave_0"), val = bool(false)]; + tensor concat_75 = concat(axis = concat_75_axis_0, interleave = concat_75_interleave_0, values = (expand_dims_112, concat_75_values1_0, var_1004, concat_75_values3_0))[name = string("concat_75")]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_74, begin_mask = model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0, end = concat_75, end_mask = model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_19_stride_0, update = key_states_37, x = coreml_update_state_45)[name = string("model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_74_write_state")]; + tensor coreml_update_state_46 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_74")]; + tensor expand_dims_114 = const()[name = string("expand_dims_114"), val = tensor([51])]; + tensor expand_dims_115 = const()[name = string("expand_dims_115"), val = tensor([0])]; + tensor expand_dims_117 = const()[name = string("expand_dims_117"), val = tensor([0])]; + tensor expand_dims_118 = const()[name = string("expand_dims_118"), val = tensor([52])]; + int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; + bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; + tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_114, expand_dims_115, current_pos, expand_dims_117))[name = string("concat_78")]; + tensor concat_79_values1_0 = const()[name = string("concat_79_values1_0"), val = tensor([0])]; + tensor concat_79_values3_0 = const()[name = string("concat_79_values3_0"), val = tensor([0])]; + int32 concat_79_axis_0 = const()[name = string("concat_79_axis_0"), val = int32(0)]; + bool concat_79_interleave_0 = const()[name = string("concat_79_interleave_0"), val = bool(false)]; + tensor concat_79 = concat(axis = concat_79_axis_0, interleave = concat_79_interleave_0, values = (expand_dims_118, concat_79_values1_0, var_1004, concat_79_values3_0))[name = string("concat_79")]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_78, begin_mask = model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0, end = concat_79, end_mask = model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_20_stride_0, update = var_5354, x = coreml_update_state_46)[name = string("model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_75_write_state")]; + tensor coreml_update_state_47 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_75")]; + tensor var_5518_begin_0 = const()[name = string("op_5518_begin_0"), val = tensor([23, 0, 0, 0])]; + tensor var_5518_end_0 = const()[name = string("op_5518_end_0"), val = tensor([24, 8, 1024, 128])]; + tensor var_5518_end_mask_0 = const()[name = string("op_5518_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_5518_cast_fp16 = slice_by_index(begin = var_5518_begin_0, end = var_5518_end_0, end_mask = var_5518_end_mask_0, x = coreml_update_state_47)[name = string("op_5518_cast_fp16")]; + tensor K_layer_cache_19_axes_0 = const()[name = string("K_layer_cache_19_axes_0"), val = tensor([0])]; + tensor K_layer_cache_19_cast_fp16 = squeeze(axes = K_layer_cache_19_axes_0, x = var_5518_cast_fp16)[name = string("K_layer_cache_19_cast_fp16")]; + tensor var_5525_begin_0 = const()[name = string("op_5525_begin_0"), val = tensor([51, 0, 0, 0])]; + tensor var_5525_end_0 = const()[name = string("op_5525_end_0"), val = tensor([52, 8, 1024, 128])]; + tensor var_5525_end_mask_0 = const()[name = string("op_5525_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_5525_cast_fp16 = slice_by_index(begin = var_5525_begin_0, end = var_5525_end_0, end_mask = var_5525_end_mask_0, x = coreml_update_state_47)[name = string("op_5525_cast_fp16")]; + tensor V_layer_cache_19_axes_0 = const()[name = string("V_layer_cache_19_axes_0"), val = tensor([0])]; + tensor V_layer_cache_19_cast_fp16 = squeeze(axes = V_layer_cache_19_axes_0, x = var_5525_cast_fp16)[name = string("V_layer_cache_19_cast_fp16")]; + tensor x_147_axes_0 = const()[name = string("x_147_axes_0"), val = tensor([1])]; + tensor x_147_cast_fp16 = expand_dims(axes = x_147_axes_0, x = K_layer_cache_19_cast_fp16)[name = string("x_147_cast_fp16")]; + tensor var_5562 = const()[name = string("op_5562"), val = tensor([1, 2, 1, 1])]; + tensor x_149_cast_fp16 = tile(reps = var_5562, x = x_147_cast_fp16)[name = string("x_149_cast_fp16")]; + tensor var_5574 = const()[name = string("op_5574"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_39_cast_fp16 = reshape(shape = var_5574, x = x_149_cast_fp16)[name = string("key_states_39_cast_fp16")]; + tensor x_153_axes_0 = const()[name = string("x_153_axes_0"), val = tensor([1])]; + tensor x_153_cast_fp16 = expand_dims(axes = x_153_axes_0, x = V_layer_cache_19_cast_fp16)[name = string("x_153_cast_fp16")]; + tensor var_5582 = const()[name = string("op_5582"), val = tensor([1, 2, 1, 1])]; + tensor x_155_cast_fp16 = tile(reps = var_5582, x = x_153_cast_fp16)[name = string("x_155_cast_fp16")]; + tensor var_5594 = const()[name = string("op_5594"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_57_cast_fp16 = reshape(shape = var_5594, x = x_155_cast_fp16)[name = string("value_states_57_cast_fp16")]; + bool var_5609_transpose_x_1 = const()[name = string("op_5609_transpose_x_1"), val = bool(false)]; + bool var_5609_transpose_y_1 = const()[name = string("op_5609_transpose_y_1"), val = bool(true)]; + tensor var_5609 = matmul(transpose_x = var_5609_transpose_x_1, transpose_y = var_5609_transpose_y_1, x = query_states_37, y = key_states_39_cast_fp16)[name = string("op_5609")]; + fp16 var_5610_to_fp16 = const()[name = string("op_5610_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_55_cast_fp16 = mul(x = var_5609, y = var_5610_to_fp16)[name = string("attn_weights_55_cast_fp16")]; + tensor attn_weights_57_cast_fp16 = add(x = attn_weights_55_cast_fp16, y = causal_mask)[name = string("attn_weights_57_cast_fp16")]; + int32 var_5645 = const()[name = string("op_5645"), val = int32(-1)]; + tensor attn_weights_59_cast_fp16 = softmax(axis = var_5645, x = attn_weights_57_cast_fp16)[name = string("attn_weights_59_cast_fp16")]; + bool attn_output_91_transpose_x_0 = const()[name = string("attn_output_91_transpose_x_0"), val = bool(false)]; + bool attn_output_91_transpose_y_0 = const()[name = string("attn_output_91_transpose_y_0"), val = bool(false)]; + tensor attn_output_91_cast_fp16 = matmul(transpose_x = attn_output_91_transpose_x_0, transpose_y = attn_output_91_transpose_y_0, x = attn_weights_59_cast_fp16, y = value_states_57_cast_fp16)[name = string("attn_output_91_cast_fp16")]; + tensor var_5656_perm_0 = const()[name = string("op_5656_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_5660 = const()[name = string("op_5660"), val = tensor([1, 1, 2048])]; + tensor var_5656_cast_fp16 = transpose(perm = var_5656_perm_0, x = attn_output_91_cast_fp16)[name = string("transpose_28")]; + tensor attn_output_95_cast_fp16 = reshape(shape = var_5660, x = var_5656_cast_fp16)[name = string("attn_output_95_cast_fp16")]; + tensor var_5665 = const()[name = string("op_5665"), val = tensor([0, 2, 1])]; + string var_5681_pad_type_0 = const()[name = string("op_5681_pad_type_0"), val = string("valid")]; + int32 var_5681_groups_0 = const()[name = string("op_5681_groups_0"), val = int32(1)]; + tensor var_5681_strides_0 = const()[name = string("op_5681_strides_0"), val = tensor([1])]; + tensor var_5681_pad_0 = const()[name = string("op_5681_pad_0"), val = tensor([0, 0])]; + tensor var_5681_dilations_0 = const()[name = string("op_5681_dilations_0"), val = tensor([1])]; + tensor squeeze_9_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702512384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706706752))))[name = string("squeeze_9_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_5666_cast_fp16 = transpose(perm = var_5665, x = attn_output_95_cast_fp16)[name = string("transpose_27")]; + tensor var_5681_cast_fp16 = conv(dilations = var_5681_dilations_0, groups = var_5681_groups_0, pad = var_5681_pad_0, pad_type = var_5681_pad_type_0, strides = var_5681_strides_0, weight = squeeze_9_cast_fp16_to_fp32_to_fp16_palettized, x = var_5666_cast_fp16)[name = string("op_5681_cast_fp16")]; + tensor var_5685 = const()[name = string("op_5685"), val = tensor([0, 2, 1])]; + tensor attn_output_99_cast_fp16 = transpose(perm = var_5685, x = var_5681_cast_fp16)[name = string("transpose_26")]; + tensor hidden_states_99_cast_fp16 = add(x = hidden_states_91_cast_fp16, y = attn_output_99_cast_fp16)[name = string("hidden_states_99_cast_fp16")]; + int32 var_5698 = const()[name = string("op_5698"), val = int32(-1)]; + fp16 const_296_promoted_to_fp16 = const()[name = string("const_296_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5700_cast_fp16 = mul(x = hidden_states_99_cast_fp16, y = const_296_promoted_to_fp16)[name = string("op_5700_cast_fp16")]; + bool input_173_interleave_0 = const()[name = string("input_173_interleave_0"), val = bool(false)]; + tensor input_173_cast_fp16 = concat(axis = var_5698, interleave = input_173_interleave_0, values = (hidden_states_99_cast_fp16, var_5700_cast_fp16))[name = string("input_173_cast_fp16")]; + tensor normed_157_axes_0 = const()[name = string("normed_157_axes_0"), val = tensor([-1])]; + fp16 var_5695_to_fp16 = const()[name = string("op_5695_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_157_cast_fp16 = layer_norm(axes = normed_157_axes_0, epsilon = var_5695_to_fp16, x = input_173_cast_fp16)[name = string("normed_157_cast_fp16")]; + tensor normed_159_begin_0 = const()[name = string("normed_159_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_159_end_0 = const()[name = string("normed_159_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_159_end_mask_0 = const()[name = string("normed_159_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_159_cast_fp16 = slice_by_index(begin = normed_159_begin_0, end = normed_159_end_0, end_mask = normed_159_end_mask_0, x = normed_157_cast_fp16)[name = string("normed_159_cast_fp16")]; + tensor const_299_promoted_to_fp16 = const()[name = string("const_299_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706837888)))]; + tensor x_157_cast_fp16 = mul(x = normed_159_cast_fp16, y = const_299_promoted_to_fp16)[name = string("x_157_cast_fp16")]; + tensor var_5725 = const()[name = string("op_5725"), val = tensor([0, 2, 1])]; + tensor input_175_axes_0 = const()[name = string("input_175_axes_0"), val = tensor([2])]; + tensor var_5726 = transpose(perm = var_5725, x = x_157_cast_fp16)[name = string("transpose_25")]; + tensor input_175 = expand_dims(axes = input_175_axes_0, x = var_5726)[name = string("input_175")]; + string input_177_pad_type_0 = const()[name = string("input_177_pad_type_0"), val = string("valid")]; + tensor input_177_strides_0 = const()[name = string("input_177_strides_0"), val = tensor([1, 1])]; + tensor input_177_pad_0 = const()[name = string("input_177_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_177_dilations_0 = const()[name = string("input_177_dilations_0"), val = tensor([1, 1])]; + int32 input_177_groups_0 = const()[name = string("input_177_groups_0"), val = int32(1)]; + tensor input_177 = conv(dilations = input_177_dilations_0, groups = input_177_groups_0, pad = input_177_pad_0, pad_type = input_177_pad_type_0, strides = input_177_strides_0, weight = model_model_layers_23_mlp_gate_proj_weight_palettized, x = input_175)[name = string("input_177")]; + string b_19_pad_type_0 = const()[name = string("b_19_pad_type_0"), val = string("valid")]; + tensor b_19_strides_0 = const()[name = string("b_19_strides_0"), val = tensor([1, 1])]; + tensor b_19_pad_0 = const()[name = string("b_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_19_dilations_0 = const()[name = string("b_19_dilations_0"), val = tensor([1, 1])]; + int32 b_19_groups_0 = const()[name = string("b_19_groups_0"), val = int32(1)]; + tensor b_19 = conv(dilations = b_19_dilations_0, groups = b_19_groups_0, pad = b_19_pad_0, pad_type = b_19_pad_type_0, strides = b_19_strides_0, weight = model_model_layers_23_mlp_up_proj_weight_palettized, x = input_175)[name = string("b_19")]; + tensor c_19 = silu(x = input_177)[name = string("c_19")]; + tensor input_179 = mul(x = c_19, y = b_19)[name = string("input_179")]; + string e_19_pad_type_0 = const()[name = string("e_19_pad_type_0"), val = string("valid")]; + tensor e_19_strides_0 = const()[name = string("e_19_strides_0"), val = tensor([1, 1])]; + tensor e_19_pad_0 = const()[name = string("e_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_19_dilations_0 = const()[name = string("e_19_dilations_0"), val = tensor([1, 1])]; + int32 e_19_groups_0 = const()[name = string("e_19_groups_0"), val = int32(1)]; + tensor e_19 = conv(dilations = e_19_dilations_0, groups = e_19_groups_0, pad = e_19_pad_0, pad_type = e_19_pad_type_0, strides = e_19_strides_0, weight = model_model_layers_23_mlp_down_proj_weight_palettized, x = input_179)[name = string("e_19")]; + tensor var_5748_axes_0 = const()[name = string("op_5748_axes_0"), val = tensor([2])]; + tensor var_5748 = squeeze(axes = var_5748_axes_0, x = e_19)[name = string("op_5748")]; + tensor var_5749 = const()[name = string("op_5749"), val = tensor([0, 2, 1])]; + tensor var_5750 = transpose(perm = var_5749, x = var_5748)[name = string("transpose_24")]; + tensor hidden_states_101_cast_fp16 = add(x = hidden_states_99_cast_fp16, y = var_5750)[name = string("hidden_states_101_cast_fp16")]; + int32 var_5762 = const()[name = string("op_5762"), val = int32(-1)]; + fp16 const_300_promoted_to_fp16 = const()[name = string("const_300_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5764_cast_fp16 = mul(x = hidden_states_101_cast_fp16, y = const_300_promoted_to_fp16)[name = string("op_5764_cast_fp16")]; + bool input_181_interleave_0 = const()[name = string("input_181_interleave_0"), val = bool(false)]; + tensor input_181_cast_fp16 = concat(axis = var_5762, interleave = input_181_interleave_0, values = (hidden_states_101_cast_fp16, var_5764_cast_fp16))[name = string("input_181_cast_fp16")]; + tensor normed_161_axes_0 = const()[name = string("normed_161_axes_0"), val = tensor([-1])]; + fp16 var_5759_to_fp16 = const()[name = string("op_5759_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_161_cast_fp16 = layer_norm(axes = normed_161_axes_0, epsilon = var_5759_to_fp16, x = input_181_cast_fp16)[name = string("normed_161_cast_fp16")]; + tensor normed_163_begin_0 = const()[name = string("normed_163_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_163_end_0 = const()[name = string("normed_163_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_163_end_mask_0 = const()[name = string("normed_163_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_163_cast_fp16 = slice_by_index(begin = normed_163_begin_0, end = normed_163_end_0, end_mask = normed_163_end_mask_0, x = normed_161_cast_fp16)[name = string("normed_163_cast_fp16")]; + tensor const_303_promoted_to_fp16 = const()[name = string("const_303_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706842048)))]; + tensor hidden_states_103_cast_fp16 = mul(x = normed_163_cast_fp16, y = const_303_promoted_to_fp16)[name = string("hidden_states_103_cast_fp16")]; + tensor var_5781 = const()[name = string("op_5781"), val = tensor([0, 2, 1])]; + tensor var_5784_axes_0 = const()[name = string("op_5784_axes_0"), val = tensor([2])]; + tensor var_5782_cast_fp16 = transpose(perm = var_5781, x = hidden_states_103_cast_fp16)[name = string("transpose_23")]; + tensor var_5784_cast_fp16 = expand_dims(axes = var_5784_axes_0, x = var_5782_cast_fp16)[name = string("op_5784_cast_fp16")]; + string var_5800_pad_type_0 = const()[name = string("op_5800_pad_type_0"), val = string("valid")]; + tensor var_5800_strides_0 = const()[name = string("op_5800_strides_0"), val = tensor([1, 1])]; + tensor var_5800_pad_0 = const()[name = string("op_5800_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5800_dilations_0 = const()[name = string("op_5800_dilations_0"), val = tensor([1, 1])]; + int32 var_5800_groups_0 = const()[name = string("op_5800_groups_0"), val = int32(1)]; + tensor var_5800 = conv(dilations = var_5800_dilations_0, groups = var_5800_groups_0, pad = var_5800_pad_0, pad_type = var_5800_pad_type_0, strides = var_5800_strides_0, weight = model_model_layers_24_self_attn_q_proj_weight_palettized, x = var_5784_cast_fp16)[name = string("op_5800")]; + tensor var_5805 = const()[name = string("op_5805"), val = tensor([1, 16, 1, 128])]; + tensor var_5806 = reshape(shape = var_5805, x = var_5800)[name = string("op_5806")]; + string var_5822_pad_type_0 = const()[name = string("op_5822_pad_type_0"), val = string("valid")]; + tensor var_5822_strides_0 = const()[name = string("op_5822_strides_0"), val = tensor([1, 1])]; + tensor var_5822_pad_0 = const()[name = string("op_5822_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5822_dilations_0 = const()[name = string("op_5822_dilations_0"), val = tensor([1, 1])]; + int32 var_5822_groups_0 = const()[name = string("op_5822_groups_0"), val = int32(1)]; + tensor var_5822 = conv(dilations = var_5822_dilations_0, groups = var_5822_groups_0, pad = var_5822_pad_0, pad_type = var_5822_pad_type_0, strides = var_5822_strides_0, weight = model_model_layers_24_self_attn_k_proj_weight_palettized, x = var_5784_cast_fp16)[name = string("op_5822")]; + tensor var_5827 = const()[name = string("op_5827"), val = tensor([1, 8, 1, 128])]; + tensor var_5828 = reshape(shape = var_5827, x = var_5822)[name = string("op_5828")]; + string var_5844_pad_type_0 = const()[name = string("op_5844_pad_type_0"), val = string("valid")]; + tensor var_5844_strides_0 = const()[name = string("op_5844_strides_0"), val = tensor([1, 1])]; + tensor var_5844_pad_0 = const()[name = string("op_5844_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_5844_dilations_0 = const()[name = string("op_5844_dilations_0"), val = tensor([1, 1])]; + int32 var_5844_groups_0 = const()[name = string("op_5844_groups_0"), val = int32(1)]; + tensor var_5844 = conv(dilations = var_5844_dilations_0, groups = var_5844_groups_0, pad = var_5844_pad_0, pad_type = var_5844_pad_type_0, strides = var_5844_strides_0, weight = model_model_layers_24_self_attn_v_proj_weight_palettized, x = var_5784_cast_fp16)[name = string("op_5844")]; + tensor var_5849 = const()[name = string("op_5849"), val = tensor([1, 8, 1, 128])]; + tensor var_5850 = reshape(shape = var_5849, x = var_5844)[name = string("op_5850")]; + int32 var_5865 = const()[name = string("op_5865"), val = int32(-1)]; + fp16 const_304_promoted = const()[name = string("const_304_promoted"), val = fp16(-0x1p+0)]; + tensor var_5867 = mul(x = var_5806, y = const_304_promoted)[name = string("op_5867")]; + bool input_185_interleave_0 = const()[name = string("input_185_interleave_0"), val = bool(false)]; + tensor input_185 = concat(axis = var_5865, interleave = input_185_interleave_0, values = (var_5806, var_5867))[name = string("input_185")]; + tensor normed_165_axes_0 = const()[name = string("normed_165_axes_0"), val = tensor([-1])]; + fp16 var_5862_to_fp16 = const()[name = string("op_5862_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_165_cast_fp16 = layer_norm(axes = normed_165_axes_0, epsilon = var_5862_to_fp16, x = input_185)[name = string("normed_165_cast_fp16")]; + tensor normed_167_begin_0 = const()[name = string("normed_167_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_167_end_0 = const()[name = string("normed_167_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_167_end_mask_0 = const()[name = string("normed_167_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_167 = slice_by_index(begin = normed_167_begin_0, end = normed_167_end_0, end_mask = normed_167_end_mask_0, x = normed_165_cast_fp16)[name = string("normed_167")]; + tensor const_307 = const()[name = string("const_307"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706846208)))]; + tensor q_21 = mul(x = normed_167, y = const_307)[name = string("q_21")]; + int32 var_5890 = const()[name = string("op_5890"), val = int32(-1)]; + fp16 const_308_promoted = const()[name = string("const_308_promoted"), val = fp16(-0x1p+0)]; + tensor var_5892 = mul(x = var_5828, y = const_308_promoted)[name = string("op_5892")]; + bool input_187_interleave_0 = const()[name = string("input_187_interleave_0"), val = bool(false)]; + tensor input_187 = concat(axis = var_5890, interleave = input_187_interleave_0, values = (var_5828, var_5892))[name = string("input_187")]; + tensor normed_169_axes_0 = const()[name = string("normed_169_axes_0"), val = tensor([-1])]; + fp16 var_5887_to_fp16 = const()[name = string("op_5887_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_169_cast_fp16 = layer_norm(axes = normed_169_axes_0, epsilon = var_5887_to_fp16, x = input_187)[name = string("normed_169_cast_fp16")]; + tensor normed_171_begin_0 = const()[name = string("normed_171_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_171_end_0 = const()[name = string("normed_171_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_171_end_mask_0 = const()[name = string("normed_171_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_171 = slice_by_index(begin = normed_171_begin_0, end = normed_171_end_0, end_mask = normed_171_end_mask_0, x = normed_169_cast_fp16)[name = string("normed_171")]; + tensor const_311 = const()[name = string("const_311"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706846528)))]; + tensor k_21 = mul(x = normed_171, y = const_311)[name = string("k_21")]; + tensor var_5906 = mul(x = q_21, y = cos_1_cast_fp16)[name = string("op_5906")]; + tensor x1_41_begin_0 = const()[name = string("x1_41_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_41_end_0 = const()[name = string("x1_41_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_41_end_mask_0 = const()[name = string("x1_41_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_41 = slice_by_index(begin = x1_41_begin_0, end = x1_41_end_0, end_mask = x1_41_end_mask_0, x = q_21)[name = string("x1_41")]; + tensor x2_41_begin_0 = const()[name = string("x2_41_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_41_end_0 = const()[name = string("x2_41_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_41_end_mask_0 = const()[name = string("x2_41_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_41 = slice_by_index(begin = x2_41_begin_0, end = x2_41_end_0, end_mask = x2_41_end_mask_0, x = q_21)[name = string("x2_41")]; + fp16 const_314_promoted = const()[name = string("const_314_promoted"), val = fp16(-0x1p+0)]; + tensor var_5927 = mul(x = x2_41, y = const_314_promoted)[name = string("op_5927")]; + int32 var_5929 = const()[name = string("op_5929"), val = int32(-1)]; + bool var_5930_interleave_0 = const()[name = string("op_5930_interleave_0"), val = bool(false)]; + tensor var_5930 = concat(axis = var_5929, interleave = var_5930_interleave_0, values = (var_5927, x1_41))[name = string("op_5930")]; + tensor var_5931 = mul(x = var_5930, y = sin_1_cast_fp16)[name = string("op_5931")]; + tensor query_states_41 = add(x = var_5906, y = var_5931)[name = string("query_states_41")]; + tensor var_5934 = mul(x = k_21, y = cos_1_cast_fp16)[name = string("op_5934")]; + tensor x1_43_begin_0 = const()[name = string("x1_43_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_43_end_0 = const()[name = string("x1_43_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_43_end_mask_0 = const()[name = string("x1_43_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_43 = slice_by_index(begin = x1_43_begin_0, end = x1_43_end_0, end_mask = x1_43_end_mask_0, x = k_21)[name = string("x1_43")]; + tensor x2_43_begin_0 = const()[name = string("x2_43_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_43_end_0 = const()[name = string("x2_43_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_43_end_mask_0 = const()[name = string("x2_43_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_43 = slice_by_index(begin = x2_43_begin_0, end = x2_43_end_0, end_mask = x2_43_end_mask_0, x = k_21)[name = string("x2_43")]; + fp16 const_317_promoted = const()[name = string("const_317_promoted"), val = fp16(-0x1p+0)]; + tensor var_5955 = mul(x = x2_43, y = const_317_promoted)[name = string("op_5955")]; + int32 var_5957 = const()[name = string("op_5957"), val = int32(-1)]; + bool var_5958_interleave_0 = const()[name = string("op_5958_interleave_0"), val = bool(false)]; + tensor var_5958 = concat(axis = var_5957, interleave = var_5958_interleave_0, values = (var_5955, x1_43))[name = string("op_5958")]; + tensor var_5959 = mul(x = var_5958, y = sin_1_cast_fp16)[name = string("op_5959")]; + tensor key_states_41 = add(x = var_5934, y = var_5959)[name = string("key_states_41")]; + tensor expand_dims_120 = const()[name = string("expand_dims_120"), val = tensor([24])]; + tensor expand_dims_121 = const()[name = string("expand_dims_121"), val = tensor([0])]; + tensor expand_dims_123 = const()[name = string("expand_dims_123"), val = tensor([0])]; + tensor expand_dims_124 = const()[name = string("expand_dims_124"), val = tensor([25])]; + int32 concat_82_axis_0 = const()[name = string("concat_82_axis_0"), val = int32(0)]; + bool concat_82_interleave_0 = const()[name = string("concat_82_interleave_0"), val = bool(false)]; + tensor concat_82 = concat(axis = concat_82_axis_0, interleave = concat_82_interleave_0, values = (expand_dims_120, expand_dims_121, current_pos, expand_dims_123))[name = string("concat_82")]; + tensor concat_83_values1_0 = const()[name = string("concat_83_values1_0"), val = tensor([0])]; + tensor concat_83_values3_0 = const()[name = string("concat_83_values3_0"), val = tensor([0])]; + int32 concat_83_axis_0 = const()[name = string("concat_83_axis_0"), val = int32(0)]; + bool concat_83_interleave_0 = const()[name = string("concat_83_interleave_0"), val = bool(false)]; + tensor concat_83 = concat(axis = concat_83_axis_0, interleave = concat_83_interleave_0, values = (expand_dims_124, concat_83_values1_0, var_1004, concat_83_values3_0))[name = string("concat_83")]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_82, begin_mask = model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0, end = concat_83, end_mask = model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_21_stride_0, update = key_states_41, x = coreml_update_state_47)[name = string("model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_76_write_state")]; + tensor coreml_update_state_48 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_76")]; + tensor expand_dims_126 = const()[name = string("expand_dims_126"), val = tensor([52])]; + tensor expand_dims_127 = const()[name = string("expand_dims_127"), val = tensor([0])]; + tensor expand_dims_129 = const()[name = string("expand_dims_129"), val = tensor([0])]; + tensor expand_dims_130 = const()[name = string("expand_dims_130"), val = tensor([53])]; + int32 concat_86_axis_0 = const()[name = string("concat_86_axis_0"), val = int32(0)]; + bool concat_86_interleave_0 = const()[name = string("concat_86_interleave_0"), val = bool(false)]; + tensor concat_86 = concat(axis = concat_86_axis_0, interleave = concat_86_interleave_0, values = (expand_dims_126, expand_dims_127, current_pos, expand_dims_129))[name = string("concat_86")]; + tensor concat_87_values1_0 = const()[name = string("concat_87_values1_0"), val = tensor([0])]; + tensor concat_87_values3_0 = const()[name = string("concat_87_values3_0"), val = tensor([0])]; + int32 concat_87_axis_0 = const()[name = string("concat_87_axis_0"), val = int32(0)]; + bool concat_87_interleave_0 = const()[name = string("concat_87_interleave_0"), val = bool(false)]; + tensor concat_87 = concat(axis = concat_87_axis_0, interleave = concat_87_interleave_0, values = (expand_dims_130, concat_87_values1_0, var_1004, concat_87_values3_0))[name = string("concat_87")]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_86, begin_mask = model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0, end = concat_87, end_mask = model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_22_stride_0, update = var_5850, x = coreml_update_state_48)[name = string("model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_77_write_state")]; + tensor coreml_update_state_49 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_77")]; + tensor var_6014_begin_0 = const()[name = string("op_6014_begin_0"), val = tensor([24, 0, 0, 0])]; + tensor var_6014_end_0 = const()[name = string("op_6014_end_0"), val = tensor([25, 8, 1024, 128])]; + tensor var_6014_end_mask_0 = const()[name = string("op_6014_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_6014_cast_fp16 = slice_by_index(begin = var_6014_begin_0, end = var_6014_end_0, end_mask = var_6014_end_mask_0, x = coreml_update_state_49)[name = string("op_6014_cast_fp16")]; + tensor K_layer_cache_21_axes_0 = const()[name = string("K_layer_cache_21_axes_0"), val = tensor([0])]; + tensor K_layer_cache_21_cast_fp16 = squeeze(axes = K_layer_cache_21_axes_0, x = var_6014_cast_fp16)[name = string("K_layer_cache_21_cast_fp16")]; + tensor var_6021_begin_0 = const()[name = string("op_6021_begin_0"), val = tensor([52, 0, 0, 0])]; + tensor var_6021_end_0 = const()[name = string("op_6021_end_0"), val = tensor([53, 8, 1024, 128])]; + tensor var_6021_end_mask_0 = const()[name = string("op_6021_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_6021_cast_fp16 = slice_by_index(begin = var_6021_begin_0, end = var_6021_end_0, end_mask = var_6021_end_mask_0, x = coreml_update_state_49)[name = string("op_6021_cast_fp16")]; + tensor V_layer_cache_21_axes_0 = const()[name = string("V_layer_cache_21_axes_0"), val = tensor([0])]; + tensor V_layer_cache_21_cast_fp16 = squeeze(axes = V_layer_cache_21_axes_0, x = var_6021_cast_fp16)[name = string("V_layer_cache_21_cast_fp16")]; + tensor x_163_axes_0 = const()[name = string("x_163_axes_0"), val = tensor([1])]; + tensor x_163_cast_fp16 = expand_dims(axes = x_163_axes_0, x = K_layer_cache_21_cast_fp16)[name = string("x_163_cast_fp16")]; + tensor var_6058 = const()[name = string("op_6058"), val = tensor([1, 2, 1, 1])]; + tensor x_165_cast_fp16 = tile(reps = var_6058, x = x_163_cast_fp16)[name = string("x_165_cast_fp16")]; + tensor var_6070 = const()[name = string("op_6070"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_43_cast_fp16 = reshape(shape = var_6070, x = x_165_cast_fp16)[name = string("key_states_43_cast_fp16")]; + tensor x_169_axes_0 = const()[name = string("x_169_axes_0"), val = tensor([1])]; + tensor x_169_cast_fp16 = expand_dims(axes = x_169_axes_0, x = V_layer_cache_21_cast_fp16)[name = string("x_169_cast_fp16")]; + tensor var_6078 = const()[name = string("op_6078"), val = tensor([1, 2, 1, 1])]; + tensor x_171_cast_fp16 = tile(reps = var_6078, x = x_169_cast_fp16)[name = string("x_171_cast_fp16")]; + tensor var_6090 = const()[name = string("op_6090"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_63_cast_fp16 = reshape(shape = var_6090, x = x_171_cast_fp16)[name = string("value_states_63_cast_fp16")]; + bool var_6105_transpose_x_1 = const()[name = string("op_6105_transpose_x_1"), val = bool(false)]; + bool var_6105_transpose_y_1 = const()[name = string("op_6105_transpose_y_1"), val = bool(true)]; + tensor var_6105 = matmul(transpose_x = var_6105_transpose_x_1, transpose_y = var_6105_transpose_y_1, x = query_states_41, y = key_states_43_cast_fp16)[name = string("op_6105")]; + fp16 var_6106_to_fp16 = const()[name = string("op_6106_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_61_cast_fp16 = mul(x = var_6105, y = var_6106_to_fp16)[name = string("attn_weights_61_cast_fp16")]; + tensor attn_weights_63_cast_fp16 = add(x = attn_weights_61_cast_fp16, y = causal_mask)[name = string("attn_weights_63_cast_fp16")]; + int32 var_6141 = const()[name = string("op_6141"), val = int32(-1)]; + tensor attn_weights_65_cast_fp16 = softmax(axis = var_6141, x = attn_weights_63_cast_fp16)[name = string("attn_weights_65_cast_fp16")]; + bool attn_output_101_transpose_x_0 = const()[name = string("attn_output_101_transpose_x_0"), val = bool(false)]; + bool attn_output_101_transpose_y_0 = const()[name = string("attn_output_101_transpose_y_0"), val = bool(false)]; + tensor attn_output_101_cast_fp16 = matmul(transpose_x = attn_output_101_transpose_x_0, transpose_y = attn_output_101_transpose_y_0, x = attn_weights_65_cast_fp16, y = value_states_63_cast_fp16)[name = string("attn_output_101_cast_fp16")]; + tensor var_6152_perm_0 = const()[name = string("op_6152_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_6156 = const()[name = string("op_6156"), val = tensor([1, 1, 2048])]; + tensor var_6152_cast_fp16 = transpose(perm = var_6152_perm_0, x = attn_output_101_cast_fp16)[name = string("transpose_22")]; + tensor attn_output_105_cast_fp16 = reshape(shape = var_6156, x = var_6152_cast_fp16)[name = string("attn_output_105_cast_fp16")]; + tensor var_6161 = const()[name = string("op_6161"), val = tensor([0, 2, 1])]; + string var_6177_pad_type_0 = const()[name = string("op_6177_pad_type_0"), val = string("valid")]; + int32 var_6177_groups_0 = const()[name = string("op_6177_groups_0"), val = int32(1)]; + tensor var_6177_strides_0 = const()[name = string("op_6177_strides_0"), val = tensor([1])]; + tensor var_6177_pad_0 = const()[name = string("op_6177_pad_0"), val = tensor([0, 0])]; + tensor var_6177_dilations_0 = const()[name = string("op_6177_dilations_0"), val = tensor([1])]; + tensor squeeze_10_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706846848))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711041216))))[name = string("squeeze_10_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_6162_cast_fp16 = transpose(perm = var_6161, x = attn_output_105_cast_fp16)[name = string("transpose_21")]; + tensor var_6177_cast_fp16 = conv(dilations = var_6177_dilations_0, groups = var_6177_groups_0, pad = var_6177_pad_0, pad_type = var_6177_pad_type_0, strides = var_6177_strides_0, weight = squeeze_10_cast_fp16_to_fp32_to_fp16_palettized, x = var_6162_cast_fp16)[name = string("op_6177_cast_fp16")]; + tensor var_6181 = const()[name = string("op_6181"), val = tensor([0, 2, 1])]; + tensor attn_output_109_cast_fp16 = transpose(perm = var_6181, x = var_6177_cast_fp16)[name = string("transpose_20")]; + tensor hidden_states_109_cast_fp16 = add(x = hidden_states_101_cast_fp16, y = attn_output_109_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; + int32 var_6194 = const()[name = string("op_6194"), val = int32(-1)]; + fp16 const_326_promoted_to_fp16 = const()[name = string("const_326_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6196_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = const_326_promoted_to_fp16)[name = string("op_6196_cast_fp16")]; + bool input_191_interleave_0 = const()[name = string("input_191_interleave_0"), val = bool(false)]; + tensor input_191_cast_fp16 = concat(axis = var_6194, interleave = input_191_interleave_0, values = (hidden_states_109_cast_fp16, var_6196_cast_fp16))[name = string("input_191_cast_fp16")]; + tensor normed_173_axes_0 = const()[name = string("normed_173_axes_0"), val = tensor([-1])]; + fp16 var_6191_to_fp16 = const()[name = string("op_6191_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_173_cast_fp16 = layer_norm(axes = normed_173_axes_0, epsilon = var_6191_to_fp16, x = input_191_cast_fp16)[name = string("normed_173_cast_fp16")]; + tensor normed_175_begin_0 = const()[name = string("normed_175_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_175_end_0 = const()[name = string("normed_175_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_175_end_mask_0 = const()[name = string("normed_175_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_175_cast_fp16 = slice_by_index(begin = normed_175_begin_0, end = normed_175_end_0, end_mask = normed_175_end_mask_0, x = normed_173_cast_fp16)[name = string("normed_175_cast_fp16")]; + tensor const_329_promoted_to_fp16 = const()[name = string("const_329_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711172352)))]; + tensor x_173_cast_fp16 = mul(x = normed_175_cast_fp16, y = const_329_promoted_to_fp16)[name = string("x_173_cast_fp16")]; + tensor var_6221 = const()[name = string("op_6221"), val = tensor([0, 2, 1])]; + tensor input_193_axes_0 = const()[name = string("input_193_axes_0"), val = tensor([2])]; + tensor var_6222 = transpose(perm = var_6221, x = x_173_cast_fp16)[name = string("transpose_19")]; + tensor input_193 = expand_dims(axes = input_193_axes_0, x = var_6222)[name = string("input_193")]; + string input_195_pad_type_0 = const()[name = string("input_195_pad_type_0"), val = string("valid")]; + tensor input_195_strides_0 = const()[name = string("input_195_strides_0"), val = tensor([1, 1])]; + tensor input_195_pad_0 = const()[name = string("input_195_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_195_dilations_0 = const()[name = string("input_195_dilations_0"), val = tensor([1, 1])]; + int32 input_195_groups_0 = const()[name = string("input_195_groups_0"), val = int32(1)]; + tensor input_195 = conv(dilations = input_195_dilations_0, groups = input_195_groups_0, pad = input_195_pad_0, pad_type = input_195_pad_type_0, strides = input_195_strides_0, weight = model_model_layers_24_mlp_gate_proj_weight_palettized, x = input_193)[name = string("input_195")]; + string b_21_pad_type_0 = const()[name = string("b_21_pad_type_0"), val = string("valid")]; + tensor b_21_strides_0 = const()[name = string("b_21_strides_0"), val = tensor([1, 1])]; + tensor b_21_pad_0 = const()[name = string("b_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_21_dilations_0 = const()[name = string("b_21_dilations_0"), val = tensor([1, 1])]; + int32 b_21_groups_0 = const()[name = string("b_21_groups_0"), val = int32(1)]; + tensor b_21 = conv(dilations = b_21_dilations_0, groups = b_21_groups_0, pad = b_21_pad_0, pad_type = b_21_pad_type_0, strides = b_21_strides_0, weight = model_model_layers_24_mlp_up_proj_weight_palettized, x = input_193)[name = string("b_21")]; + tensor c_21 = silu(x = input_195)[name = string("c_21")]; + tensor input_197 = mul(x = c_21, y = b_21)[name = string("input_197")]; + string e_21_pad_type_0 = const()[name = string("e_21_pad_type_0"), val = string("valid")]; + tensor e_21_strides_0 = const()[name = string("e_21_strides_0"), val = tensor([1, 1])]; + tensor e_21_pad_0 = const()[name = string("e_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_21_dilations_0 = const()[name = string("e_21_dilations_0"), val = tensor([1, 1])]; + int32 e_21_groups_0 = const()[name = string("e_21_groups_0"), val = int32(1)]; + tensor e_21 = conv(dilations = e_21_dilations_0, groups = e_21_groups_0, pad = e_21_pad_0, pad_type = e_21_pad_type_0, strides = e_21_strides_0, weight = model_model_layers_24_mlp_down_proj_weight_palettized, x = input_197)[name = string("e_21")]; + tensor var_6244_axes_0 = const()[name = string("op_6244_axes_0"), val = tensor([2])]; + tensor var_6244 = squeeze(axes = var_6244_axes_0, x = e_21)[name = string("op_6244")]; + tensor var_6245 = const()[name = string("op_6245"), val = tensor([0, 2, 1])]; + tensor var_6246 = transpose(perm = var_6245, x = var_6244)[name = string("transpose_18")]; + tensor hidden_states_111_cast_fp16 = add(x = hidden_states_109_cast_fp16, y = var_6246)[name = string("hidden_states_111_cast_fp16")]; + int32 var_6258 = const()[name = string("op_6258"), val = int32(-1)]; + fp16 const_330_promoted_to_fp16 = const()[name = string("const_330_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6260_cast_fp16 = mul(x = hidden_states_111_cast_fp16, y = const_330_promoted_to_fp16)[name = string("op_6260_cast_fp16")]; + bool input_199_interleave_0 = const()[name = string("input_199_interleave_0"), val = bool(false)]; + tensor input_199_cast_fp16 = concat(axis = var_6258, interleave = input_199_interleave_0, values = (hidden_states_111_cast_fp16, var_6260_cast_fp16))[name = string("input_199_cast_fp16")]; + tensor normed_177_axes_0 = const()[name = string("normed_177_axes_0"), val = tensor([-1])]; + fp16 var_6255_to_fp16 = const()[name = string("op_6255_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_177_cast_fp16 = layer_norm(axes = normed_177_axes_0, epsilon = var_6255_to_fp16, x = input_199_cast_fp16)[name = string("normed_177_cast_fp16")]; + tensor normed_179_begin_0 = const()[name = string("normed_179_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_179_end_0 = const()[name = string("normed_179_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_179_end_mask_0 = const()[name = string("normed_179_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_179_cast_fp16 = slice_by_index(begin = normed_179_begin_0, end = normed_179_end_0, end_mask = normed_179_end_mask_0, x = normed_177_cast_fp16)[name = string("normed_179_cast_fp16")]; + tensor const_333_promoted_to_fp16 = const()[name = string("const_333_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711176512)))]; + tensor hidden_states_113_cast_fp16 = mul(x = normed_179_cast_fp16, y = const_333_promoted_to_fp16)[name = string("hidden_states_113_cast_fp16")]; + tensor var_6277 = const()[name = string("op_6277"), val = tensor([0, 2, 1])]; + tensor var_6280_axes_0 = const()[name = string("op_6280_axes_0"), val = tensor([2])]; + tensor var_6278_cast_fp16 = transpose(perm = var_6277, x = hidden_states_113_cast_fp16)[name = string("transpose_17")]; + tensor var_6280_cast_fp16 = expand_dims(axes = var_6280_axes_0, x = var_6278_cast_fp16)[name = string("op_6280_cast_fp16")]; + string var_6296_pad_type_0 = const()[name = string("op_6296_pad_type_0"), val = string("valid")]; + tensor var_6296_strides_0 = const()[name = string("op_6296_strides_0"), val = tensor([1, 1])]; + tensor var_6296_pad_0 = const()[name = string("op_6296_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6296_dilations_0 = const()[name = string("op_6296_dilations_0"), val = tensor([1, 1])]; + int32 var_6296_groups_0 = const()[name = string("op_6296_groups_0"), val = int32(1)]; + tensor var_6296 = conv(dilations = var_6296_dilations_0, groups = var_6296_groups_0, pad = var_6296_pad_0, pad_type = var_6296_pad_type_0, strides = var_6296_strides_0, weight = model_model_layers_25_self_attn_q_proj_weight_palettized, x = var_6280_cast_fp16)[name = string("op_6296")]; + tensor var_6301 = const()[name = string("op_6301"), val = tensor([1, 16, 1, 128])]; + tensor var_6302 = reshape(shape = var_6301, x = var_6296)[name = string("op_6302")]; + string var_6318_pad_type_0 = const()[name = string("op_6318_pad_type_0"), val = string("valid")]; + tensor var_6318_strides_0 = const()[name = string("op_6318_strides_0"), val = tensor([1, 1])]; + tensor var_6318_pad_0 = const()[name = string("op_6318_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6318_dilations_0 = const()[name = string("op_6318_dilations_0"), val = tensor([1, 1])]; + int32 var_6318_groups_0 = const()[name = string("op_6318_groups_0"), val = int32(1)]; + tensor var_6318 = conv(dilations = var_6318_dilations_0, groups = var_6318_groups_0, pad = var_6318_pad_0, pad_type = var_6318_pad_type_0, strides = var_6318_strides_0, weight = model_model_layers_25_self_attn_k_proj_weight_palettized, x = var_6280_cast_fp16)[name = string("op_6318")]; + tensor var_6323 = const()[name = string("op_6323"), val = tensor([1, 8, 1, 128])]; + tensor var_6324 = reshape(shape = var_6323, x = var_6318)[name = string("op_6324")]; + string var_6340_pad_type_0 = const()[name = string("op_6340_pad_type_0"), val = string("valid")]; + tensor var_6340_strides_0 = const()[name = string("op_6340_strides_0"), val = tensor([1, 1])]; + tensor var_6340_pad_0 = const()[name = string("op_6340_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6340_dilations_0 = const()[name = string("op_6340_dilations_0"), val = tensor([1, 1])]; + int32 var_6340_groups_0 = const()[name = string("op_6340_groups_0"), val = int32(1)]; + tensor var_6340 = conv(dilations = var_6340_dilations_0, groups = var_6340_groups_0, pad = var_6340_pad_0, pad_type = var_6340_pad_type_0, strides = var_6340_strides_0, weight = model_model_layers_25_self_attn_v_proj_weight_palettized, x = var_6280_cast_fp16)[name = string("op_6340")]; + tensor var_6345 = const()[name = string("op_6345"), val = tensor([1, 8, 1, 128])]; + tensor var_6346 = reshape(shape = var_6345, x = var_6340)[name = string("op_6346")]; + int32 var_6361 = const()[name = string("op_6361"), val = int32(-1)]; + fp16 const_334_promoted = const()[name = string("const_334_promoted"), val = fp16(-0x1p+0)]; + tensor var_6363 = mul(x = var_6302, y = const_334_promoted)[name = string("op_6363")]; + bool input_203_interleave_0 = const()[name = string("input_203_interleave_0"), val = bool(false)]; + tensor input_203 = concat(axis = var_6361, interleave = input_203_interleave_0, values = (var_6302, var_6363))[name = string("input_203")]; + tensor normed_181_axes_0 = const()[name = string("normed_181_axes_0"), val = tensor([-1])]; + fp16 var_6358_to_fp16 = const()[name = string("op_6358_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_181_cast_fp16 = layer_norm(axes = normed_181_axes_0, epsilon = var_6358_to_fp16, x = input_203)[name = string("normed_181_cast_fp16")]; + tensor normed_183_begin_0 = const()[name = string("normed_183_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_183_end_0 = const()[name = string("normed_183_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_183_end_mask_0 = const()[name = string("normed_183_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_183 = slice_by_index(begin = normed_183_begin_0, end = normed_183_end_0, end_mask = normed_183_end_mask_0, x = normed_181_cast_fp16)[name = string("normed_183")]; + tensor const_337 = const()[name = string("const_337"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711180672)))]; + tensor q_23 = mul(x = normed_183, y = const_337)[name = string("q_23")]; + int32 var_6386 = const()[name = string("op_6386"), val = int32(-1)]; + fp16 const_338_promoted = const()[name = string("const_338_promoted"), val = fp16(-0x1p+0)]; + tensor var_6388 = mul(x = var_6324, y = const_338_promoted)[name = string("op_6388")]; + bool input_205_interleave_0 = const()[name = string("input_205_interleave_0"), val = bool(false)]; + tensor input_205 = concat(axis = var_6386, interleave = input_205_interleave_0, values = (var_6324, var_6388))[name = string("input_205")]; + tensor normed_185_axes_0 = const()[name = string("normed_185_axes_0"), val = tensor([-1])]; + fp16 var_6383_to_fp16 = const()[name = string("op_6383_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_185_cast_fp16 = layer_norm(axes = normed_185_axes_0, epsilon = var_6383_to_fp16, x = input_205)[name = string("normed_185_cast_fp16")]; + tensor normed_187_begin_0 = const()[name = string("normed_187_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_187_end_0 = const()[name = string("normed_187_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_187_end_mask_0 = const()[name = string("normed_187_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_187 = slice_by_index(begin = normed_187_begin_0, end = normed_187_end_0, end_mask = normed_187_end_mask_0, x = normed_185_cast_fp16)[name = string("normed_187")]; + tensor const_341 = const()[name = string("const_341"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711180992)))]; + tensor k_23 = mul(x = normed_187, y = const_341)[name = string("k_23")]; + tensor var_6402 = mul(x = q_23, y = cos_1_cast_fp16)[name = string("op_6402")]; + tensor x1_45_begin_0 = const()[name = string("x1_45_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_45_end_0 = const()[name = string("x1_45_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_45_end_mask_0 = const()[name = string("x1_45_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_45 = slice_by_index(begin = x1_45_begin_0, end = x1_45_end_0, end_mask = x1_45_end_mask_0, x = q_23)[name = string("x1_45")]; + tensor x2_45_begin_0 = const()[name = string("x2_45_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_45_end_0 = const()[name = string("x2_45_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_45_end_mask_0 = const()[name = string("x2_45_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_45 = slice_by_index(begin = x2_45_begin_0, end = x2_45_end_0, end_mask = x2_45_end_mask_0, x = q_23)[name = string("x2_45")]; + fp16 const_344_promoted = const()[name = string("const_344_promoted"), val = fp16(-0x1p+0)]; + tensor var_6423 = mul(x = x2_45, y = const_344_promoted)[name = string("op_6423")]; + int32 var_6425 = const()[name = string("op_6425"), val = int32(-1)]; + bool var_6426_interleave_0 = const()[name = string("op_6426_interleave_0"), val = bool(false)]; + tensor var_6426 = concat(axis = var_6425, interleave = var_6426_interleave_0, values = (var_6423, x1_45))[name = string("op_6426")]; + tensor var_6427 = mul(x = var_6426, y = sin_1_cast_fp16)[name = string("op_6427")]; + tensor query_states_45 = add(x = var_6402, y = var_6427)[name = string("query_states_45")]; + tensor var_6430 = mul(x = k_23, y = cos_1_cast_fp16)[name = string("op_6430")]; + tensor x1_47_begin_0 = const()[name = string("x1_47_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_47_end_0 = const()[name = string("x1_47_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_47_end_mask_0 = const()[name = string("x1_47_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_47 = slice_by_index(begin = x1_47_begin_0, end = x1_47_end_0, end_mask = x1_47_end_mask_0, x = k_23)[name = string("x1_47")]; + tensor x2_47_begin_0 = const()[name = string("x2_47_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_47_end_0 = const()[name = string("x2_47_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_47_end_mask_0 = const()[name = string("x2_47_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_47 = slice_by_index(begin = x2_47_begin_0, end = x2_47_end_0, end_mask = x2_47_end_mask_0, x = k_23)[name = string("x2_47")]; + fp16 const_347_promoted = const()[name = string("const_347_promoted"), val = fp16(-0x1p+0)]; + tensor var_6451 = mul(x = x2_47, y = const_347_promoted)[name = string("op_6451")]; + int32 var_6453 = const()[name = string("op_6453"), val = int32(-1)]; + bool var_6454_interleave_0 = const()[name = string("op_6454_interleave_0"), val = bool(false)]; + tensor var_6454 = concat(axis = var_6453, interleave = var_6454_interleave_0, values = (var_6451, x1_47))[name = string("op_6454")]; + tensor var_6455 = mul(x = var_6454, y = sin_1_cast_fp16)[name = string("op_6455")]; + tensor key_states_45 = add(x = var_6430, y = var_6455)[name = string("key_states_45")]; + tensor expand_dims_132 = const()[name = string("expand_dims_132"), val = tensor([25])]; + tensor expand_dims_133 = const()[name = string("expand_dims_133"), val = tensor([0])]; + tensor expand_dims_135 = const()[name = string("expand_dims_135"), val = tensor([0])]; + tensor expand_dims_136 = const()[name = string("expand_dims_136"), val = tensor([26])]; + int32 concat_90_axis_0 = const()[name = string("concat_90_axis_0"), val = int32(0)]; + bool concat_90_interleave_0 = const()[name = string("concat_90_interleave_0"), val = bool(false)]; + tensor concat_90 = concat(axis = concat_90_axis_0, interleave = concat_90_interleave_0, values = (expand_dims_132, expand_dims_133, current_pos, expand_dims_135))[name = string("concat_90")]; + tensor concat_91_values1_0 = const()[name = string("concat_91_values1_0"), val = tensor([0])]; + tensor concat_91_values3_0 = const()[name = string("concat_91_values3_0"), val = tensor([0])]; + int32 concat_91_axis_0 = const()[name = string("concat_91_axis_0"), val = int32(0)]; + bool concat_91_interleave_0 = const()[name = string("concat_91_interleave_0"), val = bool(false)]; + tensor concat_91 = concat(axis = concat_91_axis_0, interleave = concat_91_interleave_0, values = (expand_dims_136, concat_91_values1_0, var_1004, concat_91_values3_0))[name = string("concat_91")]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_90, begin_mask = model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0, end = concat_91, end_mask = model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_23_stride_0, update = key_states_45, x = coreml_update_state_49)[name = string("model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_78_write_state")]; + tensor coreml_update_state_50 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_78")]; + tensor expand_dims_138 = const()[name = string("expand_dims_138"), val = tensor([53])]; + tensor expand_dims_139 = const()[name = string("expand_dims_139"), val = tensor([0])]; + tensor expand_dims_141 = const()[name = string("expand_dims_141"), val = tensor([0])]; + tensor expand_dims_142 = const()[name = string("expand_dims_142"), val = tensor([54])]; + int32 concat_94_axis_0 = const()[name = string("concat_94_axis_0"), val = int32(0)]; + bool concat_94_interleave_0 = const()[name = string("concat_94_interleave_0"), val = bool(false)]; + tensor concat_94 = concat(axis = concat_94_axis_0, interleave = concat_94_interleave_0, values = (expand_dims_138, expand_dims_139, current_pos, expand_dims_141))[name = string("concat_94")]; + tensor concat_95_values1_0 = const()[name = string("concat_95_values1_0"), val = tensor([0])]; + tensor concat_95_values3_0 = const()[name = string("concat_95_values3_0"), val = tensor([0])]; + int32 concat_95_axis_0 = const()[name = string("concat_95_axis_0"), val = int32(0)]; + bool concat_95_interleave_0 = const()[name = string("concat_95_interleave_0"), val = bool(false)]; + tensor concat_95 = concat(axis = concat_95_axis_0, interleave = concat_95_interleave_0, values = (expand_dims_142, concat_95_values1_0, var_1004, concat_95_values3_0))[name = string("concat_95")]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_94, begin_mask = model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0, end = concat_95, end_mask = model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_24_stride_0, update = var_6346, x = coreml_update_state_50)[name = string("model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_79_write_state")]; + tensor coreml_update_state_51 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_79")]; + tensor var_6510_begin_0 = const()[name = string("op_6510_begin_0"), val = tensor([25, 0, 0, 0])]; + tensor var_6510_end_0 = const()[name = string("op_6510_end_0"), val = tensor([26, 8, 1024, 128])]; + tensor var_6510_end_mask_0 = const()[name = string("op_6510_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_6510_cast_fp16 = slice_by_index(begin = var_6510_begin_0, end = var_6510_end_0, end_mask = var_6510_end_mask_0, x = coreml_update_state_51)[name = string("op_6510_cast_fp16")]; + tensor K_layer_cache_23_axes_0 = const()[name = string("K_layer_cache_23_axes_0"), val = tensor([0])]; + tensor K_layer_cache_23_cast_fp16 = squeeze(axes = K_layer_cache_23_axes_0, x = var_6510_cast_fp16)[name = string("K_layer_cache_23_cast_fp16")]; + tensor var_6517_begin_0 = const()[name = string("op_6517_begin_0"), val = tensor([53, 0, 0, 0])]; + tensor var_6517_end_0 = const()[name = string("op_6517_end_0"), val = tensor([54, 8, 1024, 128])]; + tensor var_6517_end_mask_0 = const()[name = string("op_6517_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_6517_cast_fp16 = slice_by_index(begin = var_6517_begin_0, end = var_6517_end_0, end_mask = var_6517_end_mask_0, x = coreml_update_state_51)[name = string("op_6517_cast_fp16")]; + tensor V_layer_cache_23_axes_0 = const()[name = string("V_layer_cache_23_axes_0"), val = tensor([0])]; + tensor V_layer_cache_23_cast_fp16 = squeeze(axes = V_layer_cache_23_axes_0, x = var_6517_cast_fp16)[name = string("V_layer_cache_23_cast_fp16")]; + tensor x_179_axes_0 = const()[name = string("x_179_axes_0"), val = tensor([1])]; + tensor x_179_cast_fp16 = expand_dims(axes = x_179_axes_0, x = K_layer_cache_23_cast_fp16)[name = string("x_179_cast_fp16")]; + tensor var_6554 = const()[name = string("op_6554"), val = tensor([1, 2, 1, 1])]; + tensor x_181_cast_fp16 = tile(reps = var_6554, x = x_179_cast_fp16)[name = string("x_181_cast_fp16")]; + tensor var_6566 = const()[name = string("op_6566"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_47_cast_fp16 = reshape(shape = var_6566, x = x_181_cast_fp16)[name = string("key_states_47_cast_fp16")]; + tensor x_185_axes_0 = const()[name = string("x_185_axes_0"), val = tensor([1])]; + tensor x_185_cast_fp16 = expand_dims(axes = x_185_axes_0, x = V_layer_cache_23_cast_fp16)[name = string("x_185_cast_fp16")]; + tensor var_6574 = const()[name = string("op_6574"), val = tensor([1, 2, 1, 1])]; + tensor x_187_cast_fp16 = tile(reps = var_6574, x = x_185_cast_fp16)[name = string("x_187_cast_fp16")]; + tensor var_6586 = const()[name = string("op_6586"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_69_cast_fp16 = reshape(shape = var_6586, x = x_187_cast_fp16)[name = string("value_states_69_cast_fp16")]; + bool var_6601_transpose_x_1 = const()[name = string("op_6601_transpose_x_1"), val = bool(false)]; + bool var_6601_transpose_y_1 = const()[name = string("op_6601_transpose_y_1"), val = bool(true)]; + tensor var_6601 = matmul(transpose_x = var_6601_transpose_x_1, transpose_y = var_6601_transpose_y_1, x = query_states_45, y = key_states_47_cast_fp16)[name = string("op_6601")]; + fp16 var_6602_to_fp16 = const()[name = string("op_6602_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_67_cast_fp16 = mul(x = var_6601, y = var_6602_to_fp16)[name = string("attn_weights_67_cast_fp16")]; + tensor attn_weights_69_cast_fp16 = add(x = attn_weights_67_cast_fp16, y = causal_mask)[name = string("attn_weights_69_cast_fp16")]; + int32 var_6637 = const()[name = string("op_6637"), val = int32(-1)]; + tensor attn_weights_71_cast_fp16 = softmax(axis = var_6637, x = attn_weights_69_cast_fp16)[name = string("attn_weights_71_cast_fp16")]; + bool attn_output_111_transpose_x_0 = const()[name = string("attn_output_111_transpose_x_0"), val = bool(false)]; + bool attn_output_111_transpose_y_0 = const()[name = string("attn_output_111_transpose_y_0"), val = bool(false)]; + tensor attn_output_111_cast_fp16 = matmul(transpose_x = attn_output_111_transpose_x_0, transpose_y = attn_output_111_transpose_y_0, x = attn_weights_71_cast_fp16, y = value_states_69_cast_fp16)[name = string("attn_output_111_cast_fp16")]; + tensor var_6648_perm_0 = const()[name = string("op_6648_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_6652 = const()[name = string("op_6652"), val = tensor([1, 1, 2048])]; + tensor var_6648_cast_fp16 = transpose(perm = var_6648_perm_0, x = attn_output_111_cast_fp16)[name = string("transpose_16")]; + tensor attn_output_115_cast_fp16 = reshape(shape = var_6652, x = var_6648_cast_fp16)[name = string("attn_output_115_cast_fp16")]; + tensor var_6657 = const()[name = string("op_6657"), val = tensor([0, 2, 1])]; + string var_6673_pad_type_0 = const()[name = string("op_6673_pad_type_0"), val = string("valid")]; + int32 var_6673_groups_0 = const()[name = string("op_6673_groups_0"), val = int32(1)]; + tensor var_6673_strides_0 = const()[name = string("op_6673_strides_0"), val = tensor([1])]; + tensor var_6673_pad_0 = const()[name = string("op_6673_pad_0"), val = tensor([0, 0])]; + tensor var_6673_dilations_0 = const()[name = string("op_6673_dilations_0"), val = tensor([1])]; + tensor squeeze_11_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711181312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715375680))))[name = string("squeeze_11_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_6658_cast_fp16 = transpose(perm = var_6657, x = attn_output_115_cast_fp16)[name = string("transpose_15")]; + tensor var_6673_cast_fp16 = conv(dilations = var_6673_dilations_0, groups = var_6673_groups_0, pad = var_6673_pad_0, pad_type = var_6673_pad_type_0, strides = var_6673_strides_0, weight = squeeze_11_cast_fp16_to_fp32_to_fp16_palettized, x = var_6658_cast_fp16)[name = string("op_6673_cast_fp16")]; + tensor var_6677 = const()[name = string("op_6677"), val = tensor([0, 2, 1])]; + tensor attn_output_119_cast_fp16 = transpose(perm = var_6677, x = var_6673_cast_fp16)[name = string("transpose_14")]; + tensor hidden_states_119_cast_fp16 = add(x = hidden_states_111_cast_fp16, y = attn_output_119_cast_fp16)[name = string("hidden_states_119_cast_fp16")]; + int32 var_6690 = const()[name = string("op_6690"), val = int32(-1)]; + fp16 const_356_promoted_to_fp16 = const()[name = string("const_356_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6692_cast_fp16 = mul(x = hidden_states_119_cast_fp16, y = const_356_promoted_to_fp16)[name = string("op_6692_cast_fp16")]; + bool input_209_interleave_0 = const()[name = string("input_209_interleave_0"), val = bool(false)]; + tensor input_209_cast_fp16 = concat(axis = var_6690, interleave = input_209_interleave_0, values = (hidden_states_119_cast_fp16, var_6692_cast_fp16))[name = string("input_209_cast_fp16")]; + tensor normed_189_axes_0 = const()[name = string("normed_189_axes_0"), val = tensor([-1])]; + fp16 var_6687_to_fp16 = const()[name = string("op_6687_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_189_cast_fp16 = layer_norm(axes = normed_189_axes_0, epsilon = var_6687_to_fp16, x = input_209_cast_fp16)[name = string("normed_189_cast_fp16")]; + tensor normed_191_begin_0 = const()[name = string("normed_191_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_191_end_0 = const()[name = string("normed_191_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_191_end_mask_0 = const()[name = string("normed_191_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_191_cast_fp16 = slice_by_index(begin = normed_191_begin_0, end = normed_191_end_0, end_mask = normed_191_end_mask_0, x = normed_189_cast_fp16)[name = string("normed_191_cast_fp16")]; + tensor const_359_promoted_to_fp16 = const()[name = string("const_359_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715506816)))]; + tensor x_189_cast_fp16 = mul(x = normed_191_cast_fp16, y = const_359_promoted_to_fp16)[name = string("x_189_cast_fp16")]; + tensor var_6717 = const()[name = string("op_6717"), val = tensor([0, 2, 1])]; + tensor input_211_axes_0 = const()[name = string("input_211_axes_0"), val = tensor([2])]; + tensor var_6718 = transpose(perm = var_6717, x = x_189_cast_fp16)[name = string("transpose_13")]; + tensor input_211 = expand_dims(axes = input_211_axes_0, x = var_6718)[name = string("input_211")]; + string input_213_pad_type_0 = const()[name = string("input_213_pad_type_0"), val = string("valid")]; + tensor input_213_strides_0 = const()[name = string("input_213_strides_0"), val = tensor([1, 1])]; + tensor input_213_pad_0 = const()[name = string("input_213_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_213_dilations_0 = const()[name = string("input_213_dilations_0"), val = tensor([1, 1])]; + int32 input_213_groups_0 = const()[name = string("input_213_groups_0"), val = int32(1)]; + tensor input_213 = conv(dilations = input_213_dilations_0, groups = input_213_groups_0, pad = input_213_pad_0, pad_type = input_213_pad_type_0, strides = input_213_strides_0, weight = model_model_layers_25_mlp_gate_proj_weight_palettized, x = input_211)[name = string("input_213")]; + string b_23_pad_type_0 = const()[name = string("b_23_pad_type_0"), val = string("valid")]; + tensor b_23_strides_0 = const()[name = string("b_23_strides_0"), val = tensor([1, 1])]; + tensor b_23_pad_0 = const()[name = string("b_23_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_23_dilations_0 = const()[name = string("b_23_dilations_0"), val = tensor([1, 1])]; + int32 b_23_groups_0 = const()[name = string("b_23_groups_0"), val = int32(1)]; + tensor b_23 = conv(dilations = b_23_dilations_0, groups = b_23_groups_0, pad = b_23_pad_0, pad_type = b_23_pad_type_0, strides = b_23_strides_0, weight = model_model_layers_25_mlp_up_proj_weight_palettized, x = input_211)[name = string("b_23")]; + tensor c_23 = silu(x = input_213)[name = string("c_23")]; + tensor input_215 = mul(x = c_23, y = b_23)[name = string("input_215")]; + string e_23_pad_type_0 = const()[name = string("e_23_pad_type_0"), val = string("valid")]; + tensor e_23_strides_0 = const()[name = string("e_23_strides_0"), val = tensor([1, 1])]; + tensor e_23_pad_0 = const()[name = string("e_23_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_23_dilations_0 = const()[name = string("e_23_dilations_0"), val = tensor([1, 1])]; + int32 e_23_groups_0 = const()[name = string("e_23_groups_0"), val = int32(1)]; + tensor e_23 = conv(dilations = e_23_dilations_0, groups = e_23_groups_0, pad = e_23_pad_0, pad_type = e_23_pad_type_0, strides = e_23_strides_0, weight = model_model_layers_25_mlp_down_proj_weight_palettized, x = input_215)[name = string("e_23")]; + tensor var_6740_axes_0 = const()[name = string("op_6740_axes_0"), val = tensor([2])]; + tensor var_6740 = squeeze(axes = var_6740_axes_0, x = e_23)[name = string("op_6740")]; + tensor var_6741 = const()[name = string("op_6741"), val = tensor([0, 2, 1])]; + tensor var_6742 = transpose(perm = var_6741, x = var_6740)[name = string("transpose_12")]; + tensor hidden_states_121_cast_fp16 = add(x = hidden_states_119_cast_fp16, y = var_6742)[name = string("hidden_states_121_cast_fp16")]; + int32 var_6754 = const()[name = string("op_6754"), val = int32(-1)]; + fp16 const_360_promoted_to_fp16 = const()[name = string("const_360_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6756_cast_fp16 = mul(x = hidden_states_121_cast_fp16, y = const_360_promoted_to_fp16)[name = string("op_6756_cast_fp16")]; + bool input_217_interleave_0 = const()[name = string("input_217_interleave_0"), val = bool(false)]; + tensor input_217_cast_fp16 = concat(axis = var_6754, interleave = input_217_interleave_0, values = (hidden_states_121_cast_fp16, var_6756_cast_fp16))[name = string("input_217_cast_fp16")]; + tensor normed_193_axes_0 = const()[name = string("normed_193_axes_0"), val = tensor([-1])]; + fp16 var_6751_to_fp16 = const()[name = string("op_6751_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_193_cast_fp16 = layer_norm(axes = normed_193_axes_0, epsilon = var_6751_to_fp16, x = input_217_cast_fp16)[name = string("normed_193_cast_fp16")]; + tensor normed_195_begin_0 = const()[name = string("normed_195_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_195_end_0 = const()[name = string("normed_195_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_195_end_mask_0 = const()[name = string("normed_195_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_195_cast_fp16 = slice_by_index(begin = normed_195_begin_0, end = normed_195_end_0, end_mask = normed_195_end_mask_0, x = normed_193_cast_fp16)[name = string("normed_195_cast_fp16")]; + tensor const_363_promoted_to_fp16 = const()[name = string("const_363_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715510976)))]; + tensor hidden_states_123_cast_fp16 = mul(x = normed_195_cast_fp16, y = const_363_promoted_to_fp16)[name = string("hidden_states_123_cast_fp16")]; + tensor var_6773 = const()[name = string("op_6773"), val = tensor([0, 2, 1])]; + tensor var_6776_axes_0 = const()[name = string("op_6776_axes_0"), val = tensor([2])]; + tensor var_6774_cast_fp16 = transpose(perm = var_6773, x = hidden_states_123_cast_fp16)[name = string("transpose_11")]; + tensor var_6776_cast_fp16 = expand_dims(axes = var_6776_axes_0, x = var_6774_cast_fp16)[name = string("op_6776_cast_fp16")]; + string var_6792_pad_type_0 = const()[name = string("op_6792_pad_type_0"), val = string("valid")]; + tensor var_6792_strides_0 = const()[name = string("op_6792_strides_0"), val = tensor([1, 1])]; + tensor var_6792_pad_0 = const()[name = string("op_6792_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6792_dilations_0 = const()[name = string("op_6792_dilations_0"), val = tensor([1, 1])]; + int32 var_6792_groups_0 = const()[name = string("op_6792_groups_0"), val = int32(1)]; + tensor var_6792 = conv(dilations = var_6792_dilations_0, groups = var_6792_groups_0, pad = var_6792_pad_0, pad_type = var_6792_pad_type_0, strides = var_6792_strides_0, weight = model_model_layers_26_self_attn_q_proj_weight_palettized, x = var_6776_cast_fp16)[name = string("op_6792")]; + tensor var_6797 = const()[name = string("op_6797"), val = tensor([1, 16, 1, 128])]; + tensor var_6798 = reshape(shape = var_6797, x = var_6792)[name = string("op_6798")]; + string var_6814_pad_type_0 = const()[name = string("op_6814_pad_type_0"), val = string("valid")]; + tensor var_6814_strides_0 = const()[name = string("op_6814_strides_0"), val = tensor([1, 1])]; + tensor var_6814_pad_0 = const()[name = string("op_6814_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6814_dilations_0 = const()[name = string("op_6814_dilations_0"), val = tensor([1, 1])]; + int32 var_6814_groups_0 = const()[name = string("op_6814_groups_0"), val = int32(1)]; + tensor var_6814 = conv(dilations = var_6814_dilations_0, groups = var_6814_groups_0, pad = var_6814_pad_0, pad_type = var_6814_pad_type_0, strides = var_6814_strides_0, weight = model_model_layers_26_self_attn_k_proj_weight_palettized, x = var_6776_cast_fp16)[name = string("op_6814")]; + tensor var_6819 = const()[name = string("op_6819"), val = tensor([1, 8, 1, 128])]; + tensor var_6820 = reshape(shape = var_6819, x = var_6814)[name = string("op_6820")]; + string var_6836_pad_type_0 = const()[name = string("op_6836_pad_type_0"), val = string("valid")]; + tensor var_6836_strides_0 = const()[name = string("op_6836_strides_0"), val = tensor([1, 1])]; + tensor var_6836_pad_0 = const()[name = string("op_6836_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_6836_dilations_0 = const()[name = string("op_6836_dilations_0"), val = tensor([1, 1])]; + int32 var_6836_groups_0 = const()[name = string("op_6836_groups_0"), val = int32(1)]; + tensor var_6836 = conv(dilations = var_6836_dilations_0, groups = var_6836_groups_0, pad = var_6836_pad_0, pad_type = var_6836_pad_type_0, strides = var_6836_strides_0, weight = model_model_layers_26_self_attn_v_proj_weight_palettized, x = var_6776_cast_fp16)[name = string("op_6836")]; + tensor var_6841 = const()[name = string("op_6841"), val = tensor([1, 8, 1, 128])]; + tensor var_6842 = reshape(shape = var_6841, x = var_6836)[name = string("op_6842")]; + int32 var_6857 = const()[name = string("op_6857"), val = int32(-1)]; + fp16 const_364_promoted = const()[name = string("const_364_promoted"), val = fp16(-0x1p+0)]; + tensor var_6859 = mul(x = var_6798, y = const_364_promoted)[name = string("op_6859")]; + bool input_221_interleave_0 = const()[name = string("input_221_interleave_0"), val = bool(false)]; + tensor input_221 = concat(axis = var_6857, interleave = input_221_interleave_0, values = (var_6798, var_6859))[name = string("input_221")]; + tensor normed_197_axes_0 = const()[name = string("normed_197_axes_0"), val = tensor([-1])]; + fp16 var_6854_to_fp16 = const()[name = string("op_6854_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_197_cast_fp16 = layer_norm(axes = normed_197_axes_0, epsilon = var_6854_to_fp16, x = input_221)[name = string("normed_197_cast_fp16")]; + tensor normed_199_begin_0 = const()[name = string("normed_199_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_199_end_0 = const()[name = string("normed_199_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_199_end_mask_0 = const()[name = string("normed_199_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_199 = slice_by_index(begin = normed_199_begin_0, end = normed_199_end_0, end_mask = normed_199_end_mask_0, x = normed_197_cast_fp16)[name = string("normed_199")]; + tensor const_367 = const()[name = string("const_367"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715515136)))]; + tensor q_25 = mul(x = normed_199, y = const_367)[name = string("q_25")]; + int32 var_6882 = const()[name = string("op_6882"), val = int32(-1)]; + fp16 const_368_promoted = const()[name = string("const_368_promoted"), val = fp16(-0x1p+0)]; + tensor var_6884 = mul(x = var_6820, y = const_368_promoted)[name = string("op_6884")]; + bool input_223_interleave_0 = const()[name = string("input_223_interleave_0"), val = bool(false)]; + tensor input_223 = concat(axis = var_6882, interleave = input_223_interleave_0, values = (var_6820, var_6884))[name = string("input_223")]; + tensor normed_201_axes_0 = const()[name = string("normed_201_axes_0"), val = tensor([-1])]; + fp16 var_6879_to_fp16 = const()[name = string("op_6879_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_201_cast_fp16 = layer_norm(axes = normed_201_axes_0, epsilon = var_6879_to_fp16, x = input_223)[name = string("normed_201_cast_fp16")]; + tensor normed_203_begin_0 = const()[name = string("normed_203_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_203_end_0 = const()[name = string("normed_203_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_203_end_mask_0 = const()[name = string("normed_203_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_203 = slice_by_index(begin = normed_203_begin_0, end = normed_203_end_0, end_mask = normed_203_end_mask_0, x = normed_201_cast_fp16)[name = string("normed_203")]; + tensor const_371 = const()[name = string("const_371"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715515456)))]; + tensor k_25 = mul(x = normed_203, y = const_371)[name = string("k_25")]; + tensor var_6898 = mul(x = q_25, y = cos_1_cast_fp16)[name = string("op_6898")]; + tensor x1_49_begin_0 = const()[name = string("x1_49_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_49_end_0 = const()[name = string("x1_49_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_49_end_mask_0 = const()[name = string("x1_49_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_49 = slice_by_index(begin = x1_49_begin_0, end = x1_49_end_0, end_mask = x1_49_end_mask_0, x = q_25)[name = string("x1_49")]; + tensor x2_49_begin_0 = const()[name = string("x2_49_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_49_end_0 = const()[name = string("x2_49_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_49_end_mask_0 = const()[name = string("x2_49_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_49 = slice_by_index(begin = x2_49_begin_0, end = x2_49_end_0, end_mask = x2_49_end_mask_0, x = q_25)[name = string("x2_49")]; + fp16 const_374_promoted = const()[name = string("const_374_promoted"), val = fp16(-0x1p+0)]; + tensor var_6919 = mul(x = x2_49, y = const_374_promoted)[name = string("op_6919")]; + int32 var_6921 = const()[name = string("op_6921"), val = int32(-1)]; + bool var_6922_interleave_0 = const()[name = string("op_6922_interleave_0"), val = bool(false)]; + tensor var_6922 = concat(axis = var_6921, interleave = var_6922_interleave_0, values = (var_6919, x1_49))[name = string("op_6922")]; + tensor var_6923 = mul(x = var_6922, y = sin_1_cast_fp16)[name = string("op_6923")]; + tensor query_states_49 = add(x = var_6898, y = var_6923)[name = string("query_states_49")]; + tensor var_6926 = mul(x = k_25, y = cos_1_cast_fp16)[name = string("op_6926")]; + tensor x1_51_begin_0 = const()[name = string("x1_51_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_51_end_0 = const()[name = string("x1_51_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_51_end_mask_0 = const()[name = string("x1_51_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_51 = slice_by_index(begin = x1_51_begin_0, end = x1_51_end_0, end_mask = x1_51_end_mask_0, x = k_25)[name = string("x1_51")]; + tensor x2_51_begin_0 = const()[name = string("x2_51_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_51_end_0 = const()[name = string("x2_51_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_51_end_mask_0 = const()[name = string("x2_51_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_51 = slice_by_index(begin = x2_51_begin_0, end = x2_51_end_0, end_mask = x2_51_end_mask_0, x = k_25)[name = string("x2_51")]; + fp16 const_377_promoted = const()[name = string("const_377_promoted"), val = fp16(-0x1p+0)]; + tensor var_6947 = mul(x = x2_51, y = const_377_promoted)[name = string("op_6947")]; + int32 var_6949 = const()[name = string("op_6949"), val = int32(-1)]; + bool var_6950_interleave_0 = const()[name = string("op_6950_interleave_0"), val = bool(false)]; + tensor var_6950 = concat(axis = var_6949, interleave = var_6950_interleave_0, values = (var_6947, x1_51))[name = string("op_6950")]; + tensor var_6951 = mul(x = var_6950, y = sin_1_cast_fp16)[name = string("op_6951")]; + tensor key_states_49 = add(x = var_6926, y = var_6951)[name = string("key_states_49")]; + tensor expand_dims_144 = const()[name = string("expand_dims_144"), val = tensor([26])]; + tensor expand_dims_145 = const()[name = string("expand_dims_145"), val = tensor([0])]; + tensor expand_dims_147 = const()[name = string("expand_dims_147"), val = tensor([0])]; + tensor expand_dims_148 = const()[name = string("expand_dims_148"), val = tensor([27])]; + int32 concat_98_axis_0 = const()[name = string("concat_98_axis_0"), val = int32(0)]; + bool concat_98_interleave_0 = const()[name = string("concat_98_interleave_0"), val = bool(false)]; + tensor concat_98 = concat(axis = concat_98_axis_0, interleave = concat_98_interleave_0, values = (expand_dims_144, expand_dims_145, current_pos, expand_dims_147))[name = string("concat_98")]; + tensor concat_99_values1_0 = const()[name = string("concat_99_values1_0"), val = tensor([0])]; + tensor concat_99_values3_0 = const()[name = string("concat_99_values3_0"), val = tensor([0])]; + int32 concat_99_axis_0 = const()[name = string("concat_99_axis_0"), val = int32(0)]; + bool concat_99_interleave_0 = const()[name = string("concat_99_interleave_0"), val = bool(false)]; + tensor concat_99 = concat(axis = concat_99_axis_0, interleave = concat_99_interleave_0, values = (expand_dims_148, concat_99_values1_0, var_1004, concat_99_values3_0))[name = string("concat_99")]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_98, begin_mask = model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0, end = concat_99, end_mask = model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_25_stride_0, update = key_states_49, x = coreml_update_state_51)[name = string("model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_80_write_state")]; + tensor coreml_update_state_52 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_80")]; + tensor expand_dims_150 = const()[name = string("expand_dims_150"), val = tensor([54])]; + tensor expand_dims_151 = const()[name = string("expand_dims_151"), val = tensor([0])]; + tensor expand_dims_153 = const()[name = string("expand_dims_153"), val = tensor([0])]; + tensor expand_dims_154 = const()[name = string("expand_dims_154"), val = tensor([55])]; + int32 concat_102_axis_0 = const()[name = string("concat_102_axis_0"), val = int32(0)]; + bool concat_102_interleave_0 = const()[name = string("concat_102_interleave_0"), val = bool(false)]; + tensor concat_102 = concat(axis = concat_102_axis_0, interleave = concat_102_interleave_0, values = (expand_dims_150, expand_dims_151, current_pos, expand_dims_153))[name = string("concat_102")]; + tensor concat_103_values1_0 = const()[name = string("concat_103_values1_0"), val = tensor([0])]; + tensor concat_103_values3_0 = const()[name = string("concat_103_values3_0"), val = tensor([0])]; + int32 concat_103_axis_0 = const()[name = string("concat_103_axis_0"), val = int32(0)]; + bool concat_103_interleave_0 = const()[name = string("concat_103_interleave_0"), val = bool(false)]; + tensor concat_103 = concat(axis = concat_103_axis_0, interleave = concat_103_interleave_0, values = (expand_dims_154, concat_103_values1_0, var_1004, concat_103_values3_0))[name = string("concat_103")]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_102, begin_mask = model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0, end = concat_103, end_mask = model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_26_stride_0, update = var_6842, x = coreml_update_state_52)[name = string("model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_81_write_state")]; + tensor coreml_update_state_53 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_81")]; + tensor var_7006_begin_0 = const()[name = string("op_7006_begin_0"), val = tensor([26, 0, 0, 0])]; + tensor var_7006_end_0 = const()[name = string("op_7006_end_0"), val = tensor([27, 8, 1024, 128])]; + tensor var_7006_end_mask_0 = const()[name = string("op_7006_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_7006_cast_fp16 = slice_by_index(begin = var_7006_begin_0, end = var_7006_end_0, end_mask = var_7006_end_mask_0, x = coreml_update_state_53)[name = string("op_7006_cast_fp16")]; + tensor K_layer_cache_25_axes_0 = const()[name = string("K_layer_cache_25_axes_0"), val = tensor([0])]; + tensor K_layer_cache_25_cast_fp16 = squeeze(axes = K_layer_cache_25_axes_0, x = var_7006_cast_fp16)[name = string("K_layer_cache_25_cast_fp16")]; + tensor var_7013_begin_0 = const()[name = string("op_7013_begin_0"), val = tensor([54, 0, 0, 0])]; + tensor var_7013_end_0 = const()[name = string("op_7013_end_0"), val = tensor([55, 8, 1024, 128])]; + tensor var_7013_end_mask_0 = const()[name = string("op_7013_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_7013_cast_fp16 = slice_by_index(begin = var_7013_begin_0, end = var_7013_end_0, end_mask = var_7013_end_mask_0, x = coreml_update_state_53)[name = string("op_7013_cast_fp16")]; + tensor V_layer_cache_25_axes_0 = const()[name = string("V_layer_cache_25_axes_0"), val = tensor([0])]; + tensor V_layer_cache_25_cast_fp16 = squeeze(axes = V_layer_cache_25_axes_0, x = var_7013_cast_fp16)[name = string("V_layer_cache_25_cast_fp16")]; + tensor x_195_axes_0 = const()[name = string("x_195_axes_0"), val = tensor([1])]; + tensor x_195_cast_fp16 = expand_dims(axes = x_195_axes_0, x = K_layer_cache_25_cast_fp16)[name = string("x_195_cast_fp16")]; + tensor var_7050 = const()[name = string("op_7050"), val = tensor([1, 2, 1, 1])]; + tensor x_197_cast_fp16 = tile(reps = var_7050, x = x_195_cast_fp16)[name = string("x_197_cast_fp16")]; + tensor var_7062 = const()[name = string("op_7062"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_51_cast_fp16 = reshape(shape = var_7062, x = x_197_cast_fp16)[name = string("key_states_51_cast_fp16")]; + tensor x_201_axes_0 = const()[name = string("x_201_axes_0"), val = tensor([1])]; + tensor x_201_cast_fp16 = expand_dims(axes = x_201_axes_0, x = V_layer_cache_25_cast_fp16)[name = string("x_201_cast_fp16")]; + tensor var_7070 = const()[name = string("op_7070"), val = tensor([1, 2, 1, 1])]; + tensor x_203_cast_fp16 = tile(reps = var_7070, x = x_201_cast_fp16)[name = string("x_203_cast_fp16")]; + tensor var_7082 = const()[name = string("op_7082"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_75_cast_fp16 = reshape(shape = var_7082, x = x_203_cast_fp16)[name = string("value_states_75_cast_fp16")]; + bool var_7097_transpose_x_1 = const()[name = string("op_7097_transpose_x_1"), val = bool(false)]; + bool var_7097_transpose_y_1 = const()[name = string("op_7097_transpose_y_1"), val = bool(true)]; + tensor var_7097 = matmul(transpose_x = var_7097_transpose_x_1, transpose_y = var_7097_transpose_y_1, x = query_states_49, y = key_states_51_cast_fp16)[name = string("op_7097")]; + fp16 var_7098_to_fp16 = const()[name = string("op_7098_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_73_cast_fp16 = mul(x = var_7097, y = var_7098_to_fp16)[name = string("attn_weights_73_cast_fp16")]; + tensor attn_weights_75_cast_fp16 = add(x = attn_weights_73_cast_fp16, y = causal_mask)[name = string("attn_weights_75_cast_fp16")]; + int32 var_7133 = const()[name = string("op_7133"), val = int32(-1)]; + tensor attn_weights_77_cast_fp16 = softmax(axis = var_7133, x = attn_weights_75_cast_fp16)[name = string("attn_weights_77_cast_fp16")]; + bool attn_output_121_transpose_x_0 = const()[name = string("attn_output_121_transpose_x_0"), val = bool(false)]; + bool attn_output_121_transpose_y_0 = const()[name = string("attn_output_121_transpose_y_0"), val = bool(false)]; + tensor attn_output_121_cast_fp16 = matmul(transpose_x = attn_output_121_transpose_x_0, transpose_y = attn_output_121_transpose_y_0, x = attn_weights_77_cast_fp16, y = value_states_75_cast_fp16)[name = string("attn_output_121_cast_fp16")]; + tensor var_7144_perm_0 = const()[name = string("op_7144_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_7148 = const()[name = string("op_7148"), val = tensor([1, 1, 2048])]; + tensor var_7144_cast_fp16 = transpose(perm = var_7144_perm_0, x = attn_output_121_cast_fp16)[name = string("transpose_10")]; + tensor attn_output_125_cast_fp16 = reshape(shape = var_7148, x = var_7144_cast_fp16)[name = string("attn_output_125_cast_fp16")]; + tensor var_7153 = const()[name = string("op_7153"), val = tensor([0, 2, 1])]; + string var_7169_pad_type_0 = const()[name = string("op_7169_pad_type_0"), val = string("valid")]; + int32 var_7169_groups_0 = const()[name = string("op_7169_groups_0"), val = int32(1)]; + tensor var_7169_strides_0 = const()[name = string("op_7169_strides_0"), val = tensor([1])]; + tensor var_7169_pad_0 = const()[name = string("op_7169_pad_0"), val = tensor([0, 0])]; + tensor var_7169_dilations_0 = const()[name = string("op_7169_dilations_0"), val = tensor([1])]; + tensor squeeze_12_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715515776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719710144))))[name = string("squeeze_12_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_7154_cast_fp16 = transpose(perm = var_7153, x = attn_output_125_cast_fp16)[name = string("transpose_9")]; + tensor var_7169_cast_fp16 = conv(dilations = var_7169_dilations_0, groups = var_7169_groups_0, pad = var_7169_pad_0, pad_type = var_7169_pad_type_0, strides = var_7169_strides_0, weight = squeeze_12_cast_fp16_to_fp32_to_fp16_palettized, x = var_7154_cast_fp16)[name = string("op_7169_cast_fp16")]; + tensor var_7173 = const()[name = string("op_7173"), val = tensor([0, 2, 1])]; + tensor attn_output_129_cast_fp16 = transpose(perm = var_7173, x = var_7169_cast_fp16)[name = string("transpose_8")]; + tensor hidden_states_129_cast_fp16 = add(x = hidden_states_121_cast_fp16, y = attn_output_129_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; + int32 var_7186 = const()[name = string("op_7186"), val = int32(-1)]; + fp16 const_386_promoted_to_fp16 = const()[name = string("const_386_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7188_cast_fp16 = mul(x = hidden_states_129_cast_fp16, y = const_386_promoted_to_fp16)[name = string("op_7188_cast_fp16")]; + bool input_227_interleave_0 = const()[name = string("input_227_interleave_0"), val = bool(false)]; + tensor input_227_cast_fp16 = concat(axis = var_7186, interleave = input_227_interleave_0, values = (hidden_states_129_cast_fp16, var_7188_cast_fp16))[name = string("input_227_cast_fp16")]; + tensor normed_205_axes_0 = const()[name = string("normed_205_axes_0"), val = tensor([-1])]; + fp16 var_7183_to_fp16 = const()[name = string("op_7183_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_205_cast_fp16 = layer_norm(axes = normed_205_axes_0, epsilon = var_7183_to_fp16, x = input_227_cast_fp16)[name = string("normed_205_cast_fp16")]; + tensor normed_207_begin_0 = const()[name = string("normed_207_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_207_end_0 = const()[name = string("normed_207_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_207_end_mask_0 = const()[name = string("normed_207_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_207_cast_fp16 = slice_by_index(begin = normed_207_begin_0, end = normed_207_end_0, end_mask = normed_207_end_mask_0, x = normed_205_cast_fp16)[name = string("normed_207_cast_fp16")]; + tensor const_389_promoted_to_fp16 = const()[name = string("const_389_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719841280)))]; + tensor x_205_cast_fp16 = mul(x = normed_207_cast_fp16, y = const_389_promoted_to_fp16)[name = string("x_205_cast_fp16")]; + tensor var_7213 = const()[name = string("op_7213"), val = tensor([0, 2, 1])]; + tensor input_229_axes_0 = const()[name = string("input_229_axes_0"), val = tensor([2])]; + tensor var_7214 = transpose(perm = var_7213, x = x_205_cast_fp16)[name = string("transpose_7")]; + tensor input_229 = expand_dims(axes = input_229_axes_0, x = var_7214)[name = string("input_229")]; + string input_231_pad_type_0 = const()[name = string("input_231_pad_type_0"), val = string("valid")]; + tensor input_231_strides_0 = const()[name = string("input_231_strides_0"), val = tensor([1, 1])]; + tensor input_231_pad_0 = const()[name = string("input_231_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_231_dilations_0 = const()[name = string("input_231_dilations_0"), val = tensor([1, 1])]; + int32 input_231_groups_0 = const()[name = string("input_231_groups_0"), val = int32(1)]; + tensor input_231 = conv(dilations = input_231_dilations_0, groups = input_231_groups_0, pad = input_231_pad_0, pad_type = input_231_pad_type_0, strides = input_231_strides_0, weight = model_model_layers_26_mlp_gate_proj_weight_palettized, x = input_229)[name = string("input_231")]; + string b_25_pad_type_0 = const()[name = string("b_25_pad_type_0"), val = string("valid")]; + tensor b_25_strides_0 = const()[name = string("b_25_strides_0"), val = tensor([1, 1])]; + tensor b_25_pad_0 = const()[name = string("b_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_25_dilations_0 = const()[name = string("b_25_dilations_0"), val = tensor([1, 1])]; + int32 b_25_groups_0 = const()[name = string("b_25_groups_0"), val = int32(1)]; + tensor b_25 = conv(dilations = b_25_dilations_0, groups = b_25_groups_0, pad = b_25_pad_0, pad_type = b_25_pad_type_0, strides = b_25_strides_0, weight = model_model_layers_26_mlp_up_proj_weight_palettized, x = input_229)[name = string("b_25")]; + tensor c_25 = silu(x = input_231)[name = string("c_25")]; + tensor input_233 = mul(x = c_25, y = b_25)[name = string("input_233")]; + string e_25_pad_type_0 = const()[name = string("e_25_pad_type_0"), val = string("valid")]; + tensor e_25_strides_0 = const()[name = string("e_25_strides_0"), val = tensor([1, 1])]; + tensor e_25_pad_0 = const()[name = string("e_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_25_dilations_0 = const()[name = string("e_25_dilations_0"), val = tensor([1, 1])]; + int32 e_25_groups_0 = const()[name = string("e_25_groups_0"), val = int32(1)]; + tensor e_25 = conv(dilations = e_25_dilations_0, groups = e_25_groups_0, pad = e_25_pad_0, pad_type = e_25_pad_type_0, strides = e_25_strides_0, weight = model_model_layers_26_mlp_down_proj_weight_palettized, x = input_233)[name = string("e_25")]; + tensor var_7236_axes_0 = const()[name = string("op_7236_axes_0"), val = tensor([2])]; + tensor var_7236 = squeeze(axes = var_7236_axes_0, x = e_25)[name = string("op_7236")]; + tensor var_7237 = const()[name = string("op_7237"), val = tensor([0, 2, 1])]; + tensor var_7238 = transpose(perm = var_7237, x = var_7236)[name = string("transpose_6")]; + tensor hidden_states_131_cast_fp16 = add(x = hidden_states_129_cast_fp16, y = var_7238)[name = string("hidden_states_131_cast_fp16")]; + int32 var_7250 = const()[name = string("op_7250"), val = int32(-1)]; + fp16 const_390_promoted_to_fp16 = const()[name = string("const_390_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7252_cast_fp16 = mul(x = hidden_states_131_cast_fp16, y = const_390_promoted_to_fp16)[name = string("op_7252_cast_fp16")]; + bool input_235_interleave_0 = const()[name = string("input_235_interleave_0"), val = bool(false)]; + tensor input_235_cast_fp16 = concat(axis = var_7250, interleave = input_235_interleave_0, values = (hidden_states_131_cast_fp16, var_7252_cast_fp16))[name = string("input_235_cast_fp16")]; + tensor normed_209_axes_0 = const()[name = string("normed_209_axes_0"), val = tensor([-1])]; + fp16 var_7247_to_fp16 = const()[name = string("op_7247_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_209_cast_fp16 = layer_norm(axes = normed_209_axes_0, epsilon = var_7247_to_fp16, x = input_235_cast_fp16)[name = string("normed_209_cast_fp16")]; + tensor normed_211_begin_0 = const()[name = string("normed_211_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_211_end_0 = const()[name = string("normed_211_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_211_end_mask_0 = const()[name = string("normed_211_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_211_cast_fp16 = slice_by_index(begin = normed_211_begin_0, end = normed_211_end_0, end_mask = normed_211_end_mask_0, x = normed_209_cast_fp16)[name = string("normed_211_cast_fp16")]; + tensor const_393_promoted_to_fp16 = const()[name = string("const_393_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719845440)))]; + tensor hidden_states_133_cast_fp16 = mul(x = normed_211_cast_fp16, y = const_393_promoted_to_fp16)[name = string("hidden_states_133_cast_fp16")]; + tensor var_7269 = const()[name = string("op_7269"), val = tensor([0, 2, 1])]; + tensor var_7272_axes_0 = const()[name = string("op_7272_axes_0"), val = tensor([2])]; + tensor var_7270_cast_fp16 = transpose(perm = var_7269, x = hidden_states_133_cast_fp16)[name = string("transpose_5")]; + tensor var_7272_cast_fp16 = expand_dims(axes = var_7272_axes_0, x = var_7270_cast_fp16)[name = string("op_7272_cast_fp16")]; + string var_7288_pad_type_0 = const()[name = string("op_7288_pad_type_0"), val = string("valid")]; + tensor var_7288_strides_0 = const()[name = string("op_7288_strides_0"), val = tensor([1, 1])]; + tensor var_7288_pad_0 = const()[name = string("op_7288_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7288_dilations_0 = const()[name = string("op_7288_dilations_0"), val = tensor([1, 1])]; + int32 var_7288_groups_0 = const()[name = string("op_7288_groups_0"), val = int32(1)]; + tensor var_7288 = conv(dilations = var_7288_dilations_0, groups = var_7288_groups_0, pad = var_7288_pad_0, pad_type = var_7288_pad_type_0, strides = var_7288_strides_0, weight = model_model_layers_27_self_attn_q_proj_weight_palettized, x = var_7272_cast_fp16)[name = string("op_7288")]; + tensor var_7293 = const()[name = string("op_7293"), val = tensor([1, 16, 1, 128])]; + tensor var_7294 = reshape(shape = var_7293, x = var_7288)[name = string("op_7294")]; + string var_7310_pad_type_0 = const()[name = string("op_7310_pad_type_0"), val = string("valid")]; + tensor var_7310_strides_0 = const()[name = string("op_7310_strides_0"), val = tensor([1, 1])]; + tensor var_7310_pad_0 = const()[name = string("op_7310_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7310_dilations_0 = const()[name = string("op_7310_dilations_0"), val = tensor([1, 1])]; + int32 var_7310_groups_0 = const()[name = string("op_7310_groups_0"), val = int32(1)]; + tensor var_7310 = conv(dilations = var_7310_dilations_0, groups = var_7310_groups_0, pad = var_7310_pad_0, pad_type = var_7310_pad_type_0, strides = var_7310_strides_0, weight = model_model_layers_27_self_attn_k_proj_weight_palettized, x = var_7272_cast_fp16)[name = string("op_7310")]; + tensor var_7315 = const()[name = string("op_7315"), val = tensor([1, 8, 1, 128])]; + tensor var_7316 = reshape(shape = var_7315, x = var_7310)[name = string("op_7316")]; + string var_7332_pad_type_0 = const()[name = string("op_7332_pad_type_0"), val = string("valid")]; + tensor var_7332_strides_0 = const()[name = string("op_7332_strides_0"), val = tensor([1, 1])]; + tensor var_7332_pad_0 = const()[name = string("op_7332_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor var_7332_dilations_0 = const()[name = string("op_7332_dilations_0"), val = tensor([1, 1])]; + int32 var_7332_groups_0 = const()[name = string("op_7332_groups_0"), val = int32(1)]; + tensor var_7332 = conv(dilations = var_7332_dilations_0, groups = var_7332_groups_0, pad = var_7332_pad_0, pad_type = var_7332_pad_type_0, strides = var_7332_strides_0, weight = model_model_layers_27_self_attn_v_proj_weight_palettized, x = var_7272_cast_fp16)[name = string("op_7332")]; + tensor var_7337 = const()[name = string("op_7337"), val = tensor([1, 8, 1, 128])]; + tensor var_7338 = reshape(shape = var_7337, x = var_7332)[name = string("op_7338")]; + int32 var_7353 = const()[name = string("op_7353"), val = int32(-1)]; + fp16 const_394_promoted = const()[name = string("const_394_promoted"), val = fp16(-0x1p+0)]; + tensor var_7355 = mul(x = var_7294, y = const_394_promoted)[name = string("op_7355")]; + bool input_239_interleave_0 = const()[name = string("input_239_interleave_0"), val = bool(false)]; + tensor input_239 = concat(axis = var_7353, interleave = input_239_interleave_0, values = (var_7294, var_7355))[name = string("input_239")]; + tensor normed_213_axes_0 = const()[name = string("normed_213_axes_0"), val = tensor([-1])]; + fp16 var_7350_to_fp16 = const()[name = string("op_7350_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_213_cast_fp16 = layer_norm(axes = normed_213_axes_0, epsilon = var_7350_to_fp16, x = input_239)[name = string("normed_213_cast_fp16")]; + tensor normed_215_begin_0 = const()[name = string("normed_215_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_215_end_0 = const()[name = string("normed_215_end_0"), val = tensor([1, 16, 1, 128])]; + tensor normed_215_end_mask_0 = const()[name = string("normed_215_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_215 = slice_by_index(begin = normed_215_begin_0, end = normed_215_end_0, end_mask = normed_215_end_mask_0, x = normed_213_cast_fp16)[name = string("normed_215")]; + tensor const_397 = const()[name = string("const_397"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719849600)))]; + tensor q = mul(x = normed_215, y = const_397)[name = string("q")]; + int32 var_7378 = const()[name = string("op_7378"), val = int32(-1)]; + fp16 const_398_promoted = const()[name = string("const_398_promoted"), val = fp16(-0x1p+0)]; + tensor var_7380 = mul(x = var_7316, y = const_398_promoted)[name = string("op_7380")]; + bool input_241_interleave_0 = const()[name = string("input_241_interleave_0"), val = bool(false)]; + tensor input_241 = concat(axis = var_7378, interleave = input_241_interleave_0, values = (var_7316, var_7380))[name = string("input_241")]; + tensor normed_217_axes_0 = const()[name = string("normed_217_axes_0"), val = tensor([-1])]; + fp16 var_7375_to_fp16 = const()[name = string("op_7375_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_217_cast_fp16 = layer_norm(axes = normed_217_axes_0, epsilon = var_7375_to_fp16, x = input_241)[name = string("normed_217_cast_fp16")]; + tensor normed_219_begin_0 = const()[name = string("normed_219_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_219_end_0 = const()[name = string("normed_219_end_0"), val = tensor([1, 8, 1, 128])]; + tensor normed_219_end_mask_0 = const()[name = string("normed_219_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_219 = slice_by_index(begin = normed_219_begin_0, end = normed_219_end_0, end_mask = normed_219_end_mask_0, x = normed_217_cast_fp16)[name = string("normed_219")]; + tensor const_401 = const()[name = string("const_401"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719849920)))]; + tensor k = mul(x = normed_219, y = const_401)[name = string("k")]; + tensor var_7394 = mul(x = q, y = cos_1_cast_fp16)[name = string("op_7394")]; + tensor x1_53_begin_0 = const()[name = string("x1_53_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_53_end_0 = const()[name = string("x1_53_end_0"), val = tensor([1, 16, 1, 64])]; + tensor x1_53_end_mask_0 = const()[name = string("x1_53_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_53 = slice_by_index(begin = x1_53_begin_0, end = x1_53_end_0, end_mask = x1_53_end_mask_0, x = q)[name = string("x1_53")]; + tensor x2_53_begin_0 = const()[name = string("x2_53_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_53_end_0 = const()[name = string("x2_53_end_0"), val = tensor([1, 16, 1, 128])]; + tensor x2_53_end_mask_0 = const()[name = string("x2_53_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_53 = slice_by_index(begin = x2_53_begin_0, end = x2_53_end_0, end_mask = x2_53_end_mask_0, x = q)[name = string("x2_53")]; + fp16 const_404_promoted = const()[name = string("const_404_promoted"), val = fp16(-0x1p+0)]; + tensor var_7415 = mul(x = x2_53, y = const_404_promoted)[name = string("op_7415")]; + int32 var_7417 = const()[name = string("op_7417"), val = int32(-1)]; + bool var_7418_interleave_0 = const()[name = string("op_7418_interleave_0"), val = bool(false)]; + tensor var_7418 = concat(axis = var_7417, interleave = var_7418_interleave_0, values = (var_7415, x1_53))[name = string("op_7418")]; + tensor var_7419 = mul(x = var_7418, y = sin_1_cast_fp16)[name = string("op_7419")]; + tensor query_states_53 = add(x = var_7394, y = var_7419)[name = string("query_states_53")]; + tensor var_7422 = mul(x = k, y = cos_1_cast_fp16)[name = string("op_7422")]; + tensor x1_begin_0 = const()[name = string("x1_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_end_0 = const()[name = string("x1_end_0"), val = tensor([1, 8, 1, 64])]; + tensor x1_end_mask_0 = const()[name = string("x1_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1 = slice_by_index(begin = x1_begin_0, end = x1_end_0, end_mask = x1_end_mask_0, x = k)[name = string("x1")]; + tensor x2_begin_0 = const()[name = string("x2_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_end_0 = const()[name = string("x2_end_0"), val = tensor([1, 8, 1, 128])]; + tensor x2_end_mask_0 = const()[name = string("x2_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2 = slice_by_index(begin = x2_begin_0, end = x2_end_0, end_mask = x2_end_mask_0, x = k)[name = string("x2")]; + fp16 const_407_promoted = const()[name = string("const_407_promoted"), val = fp16(-0x1p+0)]; + tensor var_7443 = mul(x = x2, y = const_407_promoted)[name = string("op_7443")]; + int32 var_7445 = const()[name = string("op_7445"), val = int32(-1)]; + bool var_7446_interleave_0 = const()[name = string("op_7446_interleave_0"), val = bool(false)]; + tensor var_7446 = concat(axis = var_7445, interleave = var_7446_interleave_0, values = (var_7443, x1))[name = string("op_7446")]; + tensor var_7447 = mul(x = var_7446, y = sin_1_cast_fp16)[name = string("op_7447")]; + tensor key_states_53 = add(x = var_7422, y = var_7447)[name = string("key_states_53")]; + tensor expand_dims_156 = const()[name = string("expand_dims_156"), val = tensor([27])]; + tensor expand_dims_157 = const()[name = string("expand_dims_157"), val = tensor([0])]; + tensor expand_dims_159 = const()[name = string("expand_dims_159"), val = tensor([0])]; + tensor expand_dims_160 = const()[name = string("expand_dims_160"), val = tensor([28])]; + int32 concat_106_axis_0 = const()[name = string("concat_106_axis_0"), val = int32(0)]; + bool concat_106_interleave_0 = const()[name = string("concat_106_interleave_0"), val = bool(false)]; + tensor concat_106 = concat(axis = concat_106_axis_0, interleave = concat_106_interleave_0, values = (expand_dims_156, expand_dims_157, current_pos, expand_dims_159))[name = string("concat_106")]; + tensor concat_107_values1_0 = const()[name = string("concat_107_values1_0"), val = tensor([0])]; + tensor concat_107_values3_0 = const()[name = string("concat_107_values3_0"), val = tensor([0])]; + int32 concat_107_axis_0 = const()[name = string("concat_107_axis_0"), val = int32(0)]; + bool concat_107_interleave_0 = const()[name = string("concat_107_interleave_0"), val = bool(false)]; + tensor concat_107 = concat(axis = concat_107_axis_0, interleave = concat_107_interleave_0, values = (expand_dims_160, concat_107_values1_0, var_1004, concat_107_values3_0))[name = string("concat_107")]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_106, begin_mask = model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0, end = concat_107, end_mask = model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_27_stride_0, update = key_states_53, x = coreml_update_state_53)[name = string("model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_82_write_state")]; + tensor coreml_update_state_54 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_82")]; + tensor expand_dims_162 = const()[name = string("expand_dims_162"), val = tensor([55])]; + tensor expand_dims_163 = const()[name = string("expand_dims_163"), val = tensor([0])]; + tensor expand_dims_165 = const()[name = string("expand_dims_165"), val = tensor([0])]; + tensor expand_dims_166 = const()[name = string("expand_dims_166"), val = tensor([56])]; + int32 concat_110_axis_0 = const()[name = string("concat_110_axis_0"), val = int32(0)]; + bool concat_110_interleave_0 = const()[name = string("concat_110_interleave_0"), val = bool(false)]; + tensor concat_110 = concat(axis = concat_110_axis_0, interleave = concat_110_interleave_0, values = (expand_dims_162, expand_dims_163, current_pos, expand_dims_165))[name = string("concat_110")]; + tensor concat_111_values1_0 = const()[name = string("concat_111_values1_0"), val = tensor([0])]; + tensor concat_111_values3_0 = const()[name = string("concat_111_values3_0"), val = tensor([0])]; + int32 concat_111_axis_0 = const()[name = string("concat_111_axis_0"), val = int32(0)]; + bool concat_111_interleave_0 = const()[name = string("concat_111_interleave_0"), val = bool(false)]; + tensor concat_111 = concat(axis = concat_111_axis_0, interleave = concat_111_interleave_0, values = (expand_dims_166, concat_111_values1_0, var_1004, concat_111_values3_0))[name = string("concat_111")]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_110, begin_mask = model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0, end = concat_111, end_mask = model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_28_stride_0, update = var_7338, x = coreml_update_state_54)[name = string("model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_83_write_state")]; + tensor coreml_update_state_55 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_83")]; + tensor var_7502_begin_0 = const()[name = string("op_7502_begin_0"), val = tensor([27, 0, 0, 0])]; + tensor var_7502_end_0 = const()[name = string("op_7502_end_0"), val = tensor([28, 8, 1024, 128])]; + tensor var_7502_end_mask_0 = const()[name = string("op_7502_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_7502_cast_fp16 = slice_by_index(begin = var_7502_begin_0, end = var_7502_end_0, end_mask = var_7502_end_mask_0, x = coreml_update_state_55)[name = string("op_7502_cast_fp16")]; + tensor K_layer_cache_axes_0 = const()[name = string("K_layer_cache_axes_0"), val = tensor([0])]; + tensor K_layer_cache_cast_fp16 = squeeze(axes = K_layer_cache_axes_0, x = var_7502_cast_fp16)[name = string("K_layer_cache_cast_fp16")]; + tensor var_7509_begin_0 = const()[name = string("op_7509_begin_0"), val = tensor([55, 0, 0, 0])]; + tensor var_7509_end_0 = const()[name = string("op_7509_end_0"), val = tensor([1, 8, 1024, 128])]; + tensor var_7509_end_mask_0 = const()[name = string("op_7509_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_7509_cast_fp16 = slice_by_index(begin = var_7509_begin_0, end = var_7509_end_0, end_mask = var_7509_end_mask_0, x = coreml_update_state_55)[name = string("op_7509_cast_fp16")]; + tensor V_layer_cache_axes_0 = const()[name = string("V_layer_cache_axes_0"), val = tensor([0])]; + tensor V_layer_cache_cast_fp16 = squeeze(axes = V_layer_cache_axes_0, x = var_7509_cast_fp16)[name = string("V_layer_cache_cast_fp16")]; + tensor x_211_axes_0 = const()[name = string("x_211_axes_0"), val = tensor([1])]; + tensor x_211_cast_fp16 = expand_dims(axes = x_211_axes_0, x = K_layer_cache_cast_fp16)[name = string("x_211_cast_fp16")]; + tensor var_7546 = const()[name = string("op_7546"), val = tensor([1, 2, 1, 1])]; + tensor x_213_cast_fp16 = tile(reps = var_7546, x = x_211_cast_fp16)[name = string("x_213_cast_fp16")]; + tensor var_7558 = const()[name = string("op_7558"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_cast_fp16 = reshape(shape = var_7558, x = x_213_cast_fp16)[name = string("key_states_cast_fp16")]; + tensor x_217_axes_0 = const()[name = string("x_217_axes_0"), val = tensor([1])]; + tensor x_217_cast_fp16 = expand_dims(axes = x_217_axes_0, x = V_layer_cache_cast_fp16)[name = string("x_217_cast_fp16")]; + tensor var_7566 = const()[name = string("op_7566"), val = tensor([1, 2, 1, 1])]; + tensor x_219_cast_fp16 = tile(reps = var_7566, x = x_217_cast_fp16)[name = string("x_219_cast_fp16")]; + tensor var_7578 = const()[name = string("op_7578"), val = tensor([1, -1, 1024, 128])]; + tensor value_states_81_cast_fp16 = reshape(shape = var_7578, x = x_219_cast_fp16)[name = string("value_states_81_cast_fp16")]; + bool var_7593_transpose_x_1 = const()[name = string("op_7593_transpose_x_1"), val = bool(false)]; + bool var_7593_transpose_y_1 = const()[name = string("op_7593_transpose_y_1"), val = bool(true)]; + tensor var_7593 = matmul(transpose_x = var_7593_transpose_x_1, transpose_y = var_7593_transpose_y_1, x = query_states_53, y = key_states_cast_fp16)[name = string("op_7593")]; + fp16 var_7594_to_fp16 = const()[name = string("op_7594_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_79_cast_fp16 = mul(x = var_7593, y = var_7594_to_fp16)[name = string("attn_weights_79_cast_fp16")]; + tensor attn_weights_81_cast_fp16 = add(x = attn_weights_79_cast_fp16, y = causal_mask)[name = string("attn_weights_81_cast_fp16")]; + int32 var_7629 = const()[name = string("op_7629"), val = int32(-1)]; + tensor attn_weights_cast_fp16 = softmax(axis = var_7629, x = attn_weights_81_cast_fp16)[name = string("attn_weights_cast_fp16")]; + bool attn_output_131_transpose_x_0 = const()[name = string("attn_output_131_transpose_x_0"), val = bool(false)]; + bool attn_output_131_transpose_y_0 = const()[name = string("attn_output_131_transpose_y_0"), val = bool(false)]; + tensor attn_output_131_cast_fp16 = matmul(transpose_x = attn_output_131_transpose_x_0, transpose_y = attn_output_131_transpose_y_0, x = attn_weights_cast_fp16, y = value_states_81_cast_fp16)[name = string("attn_output_131_cast_fp16")]; + tensor var_7640_perm_0 = const()[name = string("op_7640_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_7644 = const()[name = string("op_7644"), val = tensor([1, 1, 2048])]; + tensor var_7640_cast_fp16 = transpose(perm = var_7640_perm_0, x = attn_output_131_cast_fp16)[name = string("transpose_4")]; + tensor attn_output_135_cast_fp16 = reshape(shape = var_7644, x = var_7640_cast_fp16)[name = string("attn_output_135_cast_fp16")]; + tensor var_7649 = const()[name = string("op_7649"), val = tensor([0, 2, 1])]; + string var_7665_pad_type_0 = const()[name = string("op_7665_pad_type_0"), val = string("valid")]; + int32 var_7665_groups_0 = const()[name = string("op_7665_groups_0"), val = int32(1)]; + tensor var_7665_strides_0 = const()[name = string("op_7665_strides_0"), val = tensor([1])]; + tensor var_7665_pad_0 = const()[name = string("op_7665_pad_0"), val = tensor([0, 0])]; + tensor var_7665_dilations_0 = const()[name = string("op_7665_dilations_0"), val = tensor([1])]; + tensor squeeze_13_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719850240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724044608))))[name = string("squeeze_13_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_7650_cast_fp16 = transpose(perm = var_7649, x = attn_output_135_cast_fp16)[name = string("transpose_3")]; + tensor var_7665_cast_fp16 = conv(dilations = var_7665_dilations_0, groups = var_7665_groups_0, pad = var_7665_pad_0, pad_type = var_7665_pad_type_0, strides = var_7665_strides_0, weight = squeeze_13_cast_fp16_to_fp32_to_fp16_palettized, x = var_7650_cast_fp16)[name = string("op_7665_cast_fp16")]; + tensor var_7669 = const()[name = string("op_7669"), val = tensor([0, 2, 1])]; + tensor attn_output_cast_fp16 = transpose(perm = var_7669, x = var_7665_cast_fp16)[name = string("transpose_2")]; + tensor hidden_states_139_cast_fp16 = add(x = hidden_states_131_cast_fp16, y = attn_output_cast_fp16)[name = string("hidden_states_139_cast_fp16")]; + int32 var_7682 = const()[name = string("op_7682"), val = int32(-1)]; + fp16 const_416_promoted_to_fp16 = const()[name = string("const_416_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7684_cast_fp16 = mul(x = hidden_states_139_cast_fp16, y = const_416_promoted_to_fp16)[name = string("op_7684_cast_fp16")]; + bool input_245_interleave_0 = const()[name = string("input_245_interleave_0"), val = bool(false)]; + tensor input_245_cast_fp16 = concat(axis = var_7682, interleave = input_245_interleave_0, values = (hidden_states_139_cast_fp16, var_7684_cast_fp16))[name = string("input_245_cast_fp16")]; + tensor normed_221_axes_0 = const()[name = string("normed_221_axes_0"), val = tensor([-1])]; + fp16 var_7679_to_fp16 = const()[name = string("op_7679_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_221_cast_fp16 = layer_norm(axes = normed_221_axes_0, epsilon = var_7679_to_fp16, x = input_245_cast_fp16)[name = string("normed_221_cast_fp16")]; + tensor normed_223_begin_0 = const()[name = string("normed_223_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_223_end_0 = const()[name = string("normed_223_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_223_end_mask_0 = const()[name = string("normed_223_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_223_cast_fp16 = slice_by_index(begin = normed_223_begin_0, end = normed_223_end_0, end_mask = normed_223_end_mask_0, x = normed_221_cast_fp16)[name = string("normed_223_cast_fp16")]; + tensor const_419_promoted_to_fp16 = const()[name = string("const_419_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724175744)))]; + tensor x_221_cast_fp16 = mul(x = normed_223_cast_fp16, y = const_419_promoted_to_fp16)[name = string("x_221_cast_fp16")]; + tensor var_7709 = const()[name = string("op_7709"), val = tensor([0, 2, 1])]; + tensor input_247_axes_0 = const()[name = string("input_247_axes_0"), val = tensor([2])]; + tensor var_7710 = transpose(perm = var_7709, x = x_221_cast_fp16)[name = string("transpose_1")]; + tensor input_247 = expand_dims(axes = input_247_axes_0, x = var_7710)[name = string("input_247")]; + string input_249_pad_type_0 = const()[name = string("input_249_pad_type_0"), val = string("valid")]; + tensor input_249_strides_0 = const()[name = string("input_249_strides_0"), val = tensor([1, 1])]; + tensor input_249_pad_0 = const()[name = string("input_249_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_249_dilations_0 = const()[name = string("input_249_dilations_0"), val = tensor([1, 1])]; + int32 input_249_groups_0 = const()[name = string("input_249_groups_0"), val = int32(1)]; + tensor input_249 = conv(dilations = input_249_dilations_0, groups = input_249_groups_0, pad = input_249_pad_0, pad_type = input_249_pad_type_0, strides = input_249_strides_0, weight = model_model_layers_27_mlp_gate_proj_weight_palettized, x = input_247)[name = string("input_249")]; + string b_pad_type_0 = const()[name = string("b_pad_type_0"), val = string("valid")]; + tensor b_strides_0 = const()[name = string("b_strides_0"), val = tensor([1, 1])]; + tensor b_pad_0 = const()[name = string("b_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_dilations_0 = const()[name = string("b_dilations_0"), val = tensor([1, 1])]; + int32 b_groups_0 = const()[name = string("b_groups_0"), val = int32(1)]; + tensor b = conv(dilations = b_dilations_0, groups = b_groups_0, pad = b_pad_0, pad_type = b_pad_type_0, strides = b_strides_0, weight = model_model_layers_27_mlp_up_proj_weight_palettized, x = input_247)[name = string("b")]; + tensor c = silu(x = input_249)[name = string("c")]; + tensor input_251 = mul(x = c, y = b)[name = string("input_251")]; + string e_pad_type_0 = const()[name = string("e_pad_type_0"), val = string("valid")]; + tensor e_strides_0 = const()[name = string("e_strides_0"), val = tensor([1, 1])]; + tensor e_pad_0 = const()[name = string("e_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_dilations_0 = const()[name = string("e_dilations_0"), val = tensor([1, 1])]; + int32 e_groups_0 = const()[name = string("e_groups_0"), val = int32(1)]; + tensor e = conv(dilations = e_dilations_0, groups = e_groups_0, pad = e_pad_0, pad_type = e_pad_type_0, strides = e_strides_0, weight = model_model_layers_27_mlp_down_proj_weight_palettized, x = input_251)[name = string("e")]; + tensor var_7732_axes_0 = const()[name = string("op_7732_axes_0"), val = tensor([2])]; + tensor var_7732 = squeeze(axes = var_7732_axes_0, x = e)[name = string("op_7732")]; + tensor var_7733 = const()[name = string("op_7733"), val = tensor([0, 2, 1])]; + tensor var_7734 = transpose(perm = var_7733, x = var_7732)[name = string("transpose_0")]; + tensor hidden_states_cast_fp16 = add(x = hidden_states_139_cast_fp16, y = var_7734)[name = string("hidden_states_cast_fp16")]; + int32 var_7746 = const()[name = string("op_7746"), val = int32(-1)]; + fp16 const_420_promoted_to_fp16 = const()[name = string("const_420_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7748_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_420_promoted_to_fp16)[name = string("op_7748_cast_fp16")]; + bool input_interleave_0 = const()[name = string("input_interleave_0"), val = bool(false)]; + tensor input_cast_fp16 = concat(axis = var_7746, interleave = input_interleave_0, values = (hidden_states_cast_fp16, var_7748_cast_fp16))[name = string("input_cast_fp16")]; + tensor normed_225_axes_0 = const()[name = string("normed_225_axes_0"), val = tensor([-1])]; + fp16 var_7743_to_fp16 = const()[name = string("op_7743_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_225_cast_fp16 = layer_norm(axes = normed_225_axes_0, epsilon = var_7743_to_fp16, x = input_cast_fp16)[name = string("normed_225_cast_fp16")]; + tensor normed_begin_0 = const()[name = string("normed_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_end_0 = const()[name = string("normed_end_0"), val = tensor([1, 1, 2048])]; + tensor normed_end_mask_0 = const()[name = string("normed_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_cast_fp16 = slice_by_index(begin = normed_begin_0, end = normed_end_0, end_mask = normed_end_mask_0, x = normed_225_cast_fp16)[name = string("normed_cast_fp16")]; + tensor const_423_promoted_to_fp16 = const()[name = string("const_423_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724179904)))]; + tensor output_hidden_states = mul(x = normed_cast_fp16, y = const_423_promoted_to_fp16)[name = string("op_7761_cast_fp16")]; + tensor position_ids_tmp = identity(x = position_ids)[name = string("position_ids_tmp")]; + } -> (output_hidden_states); + func prefill(tensor causal_mask, tensor current_pos, tensor hidden_states, state> model_model_kv_cache_0, tensor position_ids) { + tensor model_model_layers_14_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4194432))))[name = string("model_model_layers_14_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_14_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(4325568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6422784))))[name = string("model_model_layers_14_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_14_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6488384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8585600))))[name = string("model_model_layers_14_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_14_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8651200))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21234176))))[name = string("model_model_layers_14_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_14_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(21627456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34210432))))[name = string("model_model_layers_14_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_14_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34603712))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47186688))))[name = string("model_model_layers_14_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_15_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(47317824))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51512192))))[name = string("model_model_layers_15_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_15_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(51643328))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53740544))))[name = string("model_model_layers_15_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_15_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(53806144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55903360))))[name = string("model_model_layers_15_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_15_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55968960))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68551936))))[name = string("model_model_layers_15_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_15_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(68945216))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(81528192))))[name = string("model_model_layers_15_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_15_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(81921472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(94504448))))[name = string("model_model_layers_15_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_16_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(94635584))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(98829952))))[name = string("model_model_layers_16_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_16_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(98961088))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101058304))))[name = string("model_model_layers_16_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_16_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101123904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103221120))))[name = string("model_model_layers_16_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_16_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103286720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115869696))))[name = string("model_model_layers_16_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_16_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116262976))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(128845952))))[name = string("model_model_layers_16_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_16_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(129239232))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141822208))))[name = string("model_model_layers_16_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_17_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141953344))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146147712))))[name = string("model_model_layers_17_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_17_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146278848))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148376064))))[name = string("model_model_layers_17_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_17_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(148441664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(150538880))))[name = string("model_model_layers_17_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_17_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(150604480))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163187456))))[name = string("model_model_layers_17_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_17_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163580736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176163712))))[name = string("model_model_layers_17_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_17_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176556992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(189139968))))[name = string("model_model_layers_17_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_18_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(189271104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193465472))))[name = string("model_model_layers_18_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_18_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193596608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(195693824))))[name = string("model_model_layers_18_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_18_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(195759424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(197856640))))[name = string("model_model_layers_18_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_18_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(197922240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210505216))))[name = string("model_model_layers_18_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_18_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(210898496))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223481472))))[name = string("model_model_layers_18_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_18_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223874752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236457728))))[name = string("model_model_layers_18_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_19_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(236588864))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240783232))))[name = string("model_model_layers_19_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_19_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(240914368))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(243011584))))[name = string("model_model_layers_19_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_19_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(243077184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245174400))))[name = string("model_model_layers_19_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_19_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(245240000))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(257822976))))[name = string("model_model_layers_19_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_19_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(258216256))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(270799232))))[name = string("model_model_layers_19_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_19_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(271192512))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283775488))))[name = string("model_model_layers_19_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_20_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(283906624))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288100992))))[name = string("model_model_layers_20_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_20_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(288232128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290329344))))[name = string("model_model_layers_20_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_20_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290394944))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292492160))))[name = string("model_model_layers_20_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_20_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(292557760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305140736))))[name = string("model_model_layers_20_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_20_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(305534016))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318116992))))[name = string("model_model_layers_20_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_20_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(318510272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331093248))))[name = string("model_model_layers_20_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_21_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(331224384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335418752))))[name = string("model_model_layers_21_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_21_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(335549888))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(337647104))))[name = string("model_model_layers_21_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_21_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(337712704))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339809920))))[name = string("model_model_layers_21_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_21_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(339875520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352458496))))[name = string("model_model_layers_21_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_21_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(352851776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(365434752))))[name = string("model_model_layers_21_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_21_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(365828032))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(378411008))))[name = string("model_model_layers_21_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_22_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(378542144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(382736512))))[name = string("model_model_layers_22_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_22_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(382867648))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(384964864))))[name = string("model_model_layers_22_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_22_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385030464))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(387127680))))[name = string("model_model_layers_22_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_22_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(387193280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(399776256))))[name = string("model_model_layers_22_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_22_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(400169536))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(412752512))))[name = string("model_model_layers_22_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_22_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(413145792))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(425728768))))[name = string("model_model_layers_22_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_23_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(425859904))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(430054272))))[name = string("model_model_layers_23_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_23_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(430185408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432282624))))[name = string("model_model_layers_23_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_23_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432348224))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(434445440))))[name = string("model_model_layers_23_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_23_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(434511040))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(447094016))))[name = string("model_model_layers_23_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_23_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(447487296))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(460070272))))[name = string("model_model_layers_23_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_23_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(460463552))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(473046528))))[name = string("model_model_layers_23_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_24_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(473177664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(477372032))))[name = string("model_model_layers_24_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_24_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(477503168))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479600384))))[name = string("model_model_layers_24_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_24_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479665984))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(481763200))))[name = string("model_model_layers_24_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_24_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(481828800))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(494411776))))[name = string("model_model_layers_24_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_24_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(494805056))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(507388032))))[name = string("model_model_layers_24_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_24_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(507781312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(520364288))))[name = string("model_model_layers_24_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_25_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(520495424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524689792))))[name = string("model_model_layers_25_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_25_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524820928))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526918144))))[name = string("model_model_layers_25_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_25_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526983744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(529080960))))[name = string("model_model_layers_25_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_25_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(529146560))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(541729536))))[name = string("model_model_layers_25_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_25_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(542122816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(554705792))))[name = string("model_model_layers_25_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_25_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(555099072))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(567682048))))[name = string("model_model_layers_25_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_26_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(567813184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(572007552))))[name = string("model_model_layers_26_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_26_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(572138688))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(574235904))))[name = string("model_model_layers_26_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_26_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(574301504))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(576398720))))[name = string("model_model_layers_26_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_26_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(576464320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(589047296))))[name = string("model_model_layers_26_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_26_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(589440576))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(602023552))))[name = string("model_model_layers_26_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_26_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(602416832))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(614999808))))[name = string("model_model_layers_26_mlp_down_proj_weight_palettized")]; + tensor model_model_layers_27_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(615130944))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(619325312))))[name = string("model_model_layers_27_self_attn_q_proj_weight_palettized")]; + tensor model_model_layers_27_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(619456448))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(621553664))))[name = string("model_model_layers_27_self_attn_k_proj_weight_palettized")]; + tensor model_model_layers_27_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(621619264))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(623716480))))[name = string("model_model_layers_27_self_attn_v_proj_weight_palettized")]; + tensor model_model_layers_27_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(623782080))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(636365056))))[name = string("model_model_layers_27_mlp_gate_proj_weight_palettized")]; + tensor model_model_layers_27_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(636758336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(649341312))))[name = string("model_model_layers_27_mlp_up_proj_weight_palettized")]; + tensor model_model_layers_27_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(649734592))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(662317568))))[name = string("model_model_layers_27_mlp_down_proj_weight_palettized")]; + int32 var_766_batch_dims_0 = const()[name = string("op_766_batch_dims_0"), val = int32(0)]; + bool var_766_validate_indices_0 = const()[name = string("op_766_validate_indices_0"), val = bool(false)]; + tensor var_758_to_fp16 = const()[name = string("op_758_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(662973056)))]; + string position_ids_to_int16_dtype_0 = const()[name = string("position_ids_to_int16_dtype_0"), val = string("int16")]; + string cast_118_dtype_0 = const()[name = string("cast_118_dtype_0"), val = string("int32")]; + int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; + tensor position_ids_to_int16 = cast(dtype = position_ids_to_int16_dtype_0, x = position_ids)[name = string("cast_5")]; + tensor cast_118 = cast(dtype = cast_118_dtype_0, x = position_ids_to_int16)[name = string("cast_4")]; + tensor greater_equal_0 = greater_equal(x = cast_118, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; + int32 slice_by_index_112 = const()[name = string("slice_by_index_112"), val = int32(2048)]; + tensor add_0 = add(x = cast_118, y = slice_by_index_112)[name = string("add_0")]; + tensor select_0 = select(a = cast_118, b = add_0, cond = greater_equal_0)[name = string("select_0")]; + string select_0_to_int16_dtype_0 = const()[name = string("select_0_to_int16_dtype_0"), val = string("int16")]; + string cast_0_dtype_0 = const()[name = string("cast_0_dtype_0"), val = string("int32")]; + int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; + tensor select_0_to_int16 = cast(dtype = select_0_to_int16_dtype_0, x = select_0)[name = string("cast_3")]; + tensor cast_0 = cast(dtype = cast_0_dtype_0, x = select_0_to_int16)[name = string("cast_2")]; + tensor greater_equal_0_1 = greater_equal(x = cast_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; + int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(2048)]; + tensor add_0_1 = add(x = cast_0, y = slice_by_index_0)[name = string("add_0_1")]; + tensor select_0_1 = select(a = cast_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; + int32 op_766_cast_fp16_cast_uint16_cast_uint16_axis_0 = const()[name = string("op_766_cast_fp16_cast_uint16_cast_uint16_axis_0"), val = int32(1)]; + tensor op_766_cast_fp16_cast_uint16_cast_uint16 = gather(axis = op_766_cast_fp16_cast_uint16_cast_uint16_axis_0, batch_dims = var_766_batch_dims_0, indices = select_0_1, validate_indices = var_766_validate_indices_0, x = var_758_to_fp16)[name = string("op_766_cast_fp16_cast_uint16_cast_uint16")]; + tensor var_770 = const()[name = string("op_770"), val = tensor([1, 128, 1, 128])]; + tensor cos_1_cast_fp16 = reshape(shape = var_770, x = op_766_cast_fp16_cast_uint16_cast_uint16)[name = string("cos_1_cast_fp16")]; + int32 var_780_axis_0 = const()[name = string("op_780_axis_0"), val = int32(1)]; + int32 var_780_batch_dims_0 = const()[name = string("op_780_batch_dims_0"), val = int32(0)]; + bool var_780_validate_indices_0 = const()[name = string("op_780_validate_indices_0"), val = bool(false)]; + tensor var_772_to_fp16 = const()[name = string("op_772_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(662448704)))]; + string position_ids_to_uint16_dtype_0 = const()[name = string("position_ids_to_uint16_dtype_0"), val = string("uint16")]; + tensor position_ids_to_uint16 = cast(dtype = position_ids_to_uint16_dtype_0, x = position_ids)[name = string("cast_1")]; + tensor var_780_cast_fp16_cast_uint16 = gather(axis = var_780_axis_0, batch_dims = var_780_batch_dims_0, indices = position_ids_to_uint16, validate_indices = var_780_validate_indices_0, x = var_772_to_fp16)[name = string("op_780_cast_fp16_cast_uint16")]; + tensor var_784 = const()[name = string("op_784"), val = tensor([1, 128, 1, 128])]; + tensor sin_1_cast_fp16 = reshape(shape = var_784, x = var_780_cast_fp16_cast_uint16)[name = string("sin_1_cast_fp16")]; + int32 var_805 = const()[name = string("op_805"), val = int32(-1)]; + fp16 const_1_promoted_to_fp16 = const()[name = string("const_1_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_807_cast_fp16 = mul(x = hidden_states, y = const_1_promoted_to_fp16)[name = string("op_807_cast_fp16")]; + bool input_1_interleave_0 = const()[name = string("input_1_interleave_0"), val = bool(false)]; + tensor input_1_cast_fp16 = concat(axis = var_805, interleave = input_1_interleave_0, values = (hidden_states, var_807_cast_fp16))[name = string("input_1_cast_fp16")]; + tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; + fp16 var_802_to_fp16 = const()[name = string("op_802_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_802_to_fp16, x = input_1_cast_fp16)[name = string("normed_1_cast_fp16")]; + tensor normed_3_begin_0 = const()[name = string("normed_3_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_3_end_0 = const()[name = string("normed_3_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_3_end_mask_0 = const()[name = string("normed_3_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_3_cast_fp16 = slice_by_index(begin = normed_3_begin_0, end = normed_3_end_0, end_mask = normed_3_end_mask_0, x = normed_1_cast_fp16)[name = string("normed_3_cast_fp16")]; + tensor const_4_promoted_to_fp16 = const()[name = string("const_4_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(663497408)))]; + tensor hidden_states_3_cast_fp16 = mul(x = normed_3_cast_fp16, y = const_4_promoted_to_fp16)[name = string("hidden_states_3_cast_fp16")]; + tensor var_830 = const()[name = string("op_830"), val = tensor([0, 2, 1])]; + tensor var_833_axes_0 = const()[name = string("op_833_axes_0"), val = tensor([2])]; + tensor var_831_cast_fp16 = transpose(perm = var_830, x = hidden_states_3_cast_fp16)[name = string("transpose_127")]; + tensor var_833_cast_fp16 = expand_dims(axes = var_833_axes_0, x = var_831_cast_fp16)[name = string("op_833_cast_fp16")]; + string query_states_1_pad_type_0 = const()[name = string("query_states_1_pad_type_0"), val = string("valid")]; + tensor query_states_1_strides_0 = const()[name = string("query_states_1_strides_0"), val = tensor([1, 1])]; + tensor query_states_1_pad_0 = const()[name = string("query_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_1_dilations_0 = const()[name = string("query_states_1_dilations_0"), val = tensor([1, 1])]; + int32 query_states_1_groups_0 = const()[name = string("query_states_1_groups_0"), val = int32(1)]; + tensor query_states_1 = conv(dilations = query_states_1_dilations_0, groups = query_states_1_groups_0, pad = query_states_1_pad_0, pad_type = query_states_1_pad_type_0, strides = query_states_1_strides_0, weight = model_model_layers_14_self_attn_q_proj_weight_palettized, x = var_833_cast_fp16)[name = string("query_states_1")]; + string key_states_1_pad_type_0 = const()[name = string("key_states_1_pad_type_0"), val = string("valid")]; + tensor key_states_1_strides_0 = const()[name = string("key_states_1_strides_0"), val = tensor([1, 1])]; + tensor key_states_1_pad_0 = const()[name = string("key_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_1_dilations_0 = const()[name = string("key_states_1_dilations_0"), val = tensor([1, 1])]; + int32 key_states_1_groups_0 = const()[name = string("key_states_1_groups_0"), val = int32(1)]; + tensor key_states_1 = conv(dilations = key_states_1_dilations_0, groups = key_states_1_groups_0, pad = key_states_1_pad_0, pad_type = key_states_1_pad_type_0, strides = key_states_1_strides_0, weight = model_model_layers_14_self_attn_k_proj_weight_palettized, x = var_833_cast_fp16)[name = string("key_states_1")]; + string value_states_1_pad_type_0 = const()[name = string("value_states_1_pad_type_0"), val = string("valid")]; + tensor value_states_1_strides_0 = const()[name = string("value_states_1_strides_0"), val = tensor([1, 1])]; + tensor value_states_1_pad_0 = const()[name = string("value_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_1_dilations_0 = const()[name = string("value_states_1_dilations_0"), val = tensor([1, 1])]; + int32 value_states_1_groups_0 = const()[name = string("value_states_1_groups_0"), val = int32(1)]; + tensor value_states_1 = conv(dilations = value_states_1_dilations_0, groups = value_states_1_groups_0, pad = value_states_1_pad_0, pad_type = value_states_1_pad_type_0, strides = value_states_1_strides_0, weight = model_model_layers_14_self_attn_v_proj_weight_palettized, x = var_833_cast_fp16)[name = string("value_states_1")]; + tensor var_875 = const()[name = string("op_875"), val = tensor([1, 16, 128, 128])]; + tensor var_876 = reshape(shape = var_875, x = query_states_1)[name = string("op_876")]; + tensor var_881 = const()[name = string("op_881"), val = tensor([0, 1, 3, 2])]; + tensor var_886 = const()[name = string("op_886"), val = tensor([1, 8, 128, 128])]; + tensor var_887 = reshape(shape = var_886, x = key_states_1)[name = string("op_887")]; + tensor var_892 = const()[name = string("op_892"), val = tensor([0, 1, 3, 2])]; + tensor var_897 = const()[name = string("op_897"), val = tensor([1, 8, 128, 128])]; + tensor var_898 = reshape(shape = var_897, x = value_states_1)[name = string("op_898")]; + tensor var_903 = const()[name = string("op_903"), val = tensor([0, 1, 3, 2])]; + int32 var_914 = const()[name = string("op_914"), val = int32(-1)]; + fp16 const_6_promoted = const()[name = string("const_6_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_5 = transpose(perm = var_881, x = var_876)[name = string("transpose_126")]; + tensor var_916 = mul(x = hidden_states_5, y = const_6_promoted)[name = string("op_916")]; + bool input_5_interleave_0 = const()[name = string("input_5_interleave_0"), val = bool(false)]; + tensor input_5 = concat(axis = var_914, interleave = input_5_interleave_0, values = (hidden_states_5, var_916))[name = string("input_5")]; + tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; + fp16 var_911_to_fp16 = const()[name = string("op_911_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_911_to_fp16, x = input_5)[name = string("normed_5_cast_fp16")]; + tensor normed_7_begin_0 = const()[name = string("normed_7_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_7_end_0 = const()[name = string("normed_7_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_7_end_mask_0 = const()[name = string("normed_7_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_7 = slice_by_index(begin = normed_7_begin_0, end = normed_7_end_0, end_mask = normed_7_end_mask_0, x = normed_5_cast_fp16)[name = string("normed_7")]; + tensor const_9 = const()[name = string("const_9"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(663501568)))]; + tensor q_1 = mul(x = normed_7, y = const_9)[name = string("q_1")]; + int32 var_939 = const()[name = string("op_939"), val = int32(-1)]; + fp16 const_10_promoted = const()[name = string("const_10_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_7 = transpose(perm = var_892, x = var_887)[name = string("transpose_125")]; + tensor var_941 = mul(x = hidden_states_7, y = const_10_promoted)[name = string("op_941")]; + bool input_7_interleave_0 = const()[name = string("input_7_interleave_0"), val = bool(false)]; + tensor input_7 = concat(axis = var_939, interleave = input_7_interleave_0, values = (hidden_states_7, var_941))[name = string("input_7")]; + tensor normed_9_axes_0 = const()[name = string("normed_9_axes_0"), val = tensor([-1])]; + fp16 var_936_to_fp16 = const()[name = string("op_936_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_9_cast_fp16 = layer_norm(axes = normed_9_axes_0, epsilon = var_936_to_fp16, x = input_7)[name = string("normed_9_cast_fp16")]; + tensor normed_11_begin_0 = const()[name = string("normed_11_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_11_end_0 = const()[name = string("normed_11_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_11_end_mask_0 = const()[name = string("normed_11_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_11 = slice_by_index(begin = normed_11_begin_0, end = normed_11_end_0, end_mask = normed_11_end_mask_0, x = normed_9_cast_fp16)[name = string("normed_11")]; + tensor const_13 = const()[name = string("const_13"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(663501888)))]; + tensor k_1 = mul(x = normed_11, y = const_13)[name = string("k_1")]; + tensor var_959 = const()[name = string("op_959"), val = tensor([0, 2, 1, 3])]; + tensor var_965 = const()[name = string("op_965"), val = tensor([0, 2, 1, 3])]; + tensor cos_5 = transpose(perm = var_959, x = cos_1_cast_fp16)[name = string("transpose_124")]; + tensor var_967 = mul(x = q_1, y = cos_5)[name = string("op_967")]; + tensor x1_1_begin_0 = const()[name = string("x1_1_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_1_end_0 = const()[name = string("x1_1_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_1_end_mask_0 = const()[name = string("x1_1_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_1 = slice_by_index(begin = x1_1_begin_0, end = x1_1_end_0, end_mask = x1_1_end_mask_0, x = q_1)[name = string("x1_1")]; + tensor x2_1_begin_0 = const()[name = string("x2_1_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_1_end_0 = const()[name = string("x2_1_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_1_end_mask_0 = const()[name = string("x2_1_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_1 = slice_by_index(begin = x2_1_begin_0, end = x2_1_end_0, end_mask = x2_1_end_mask_0, x = q_1)[name = string("x2_1")]; + fp16 const_16_promoted = const()[name = string("const_16_promoted"), val = fp16(-0x1p+0)]; + tensor var_988 = mul(x = x2_1, y = const_16_promoted)[name = string("op_988")]; + int32 var_990 = const()[name = string("op_990"), val = int32(-1)]; + bool var_991_interleave_0 = const()[name = string("op_991_interleave_0"), val = bool(false)]; + tensor var_991 = concat(axis = var_990, interleave = var_991_interleave_0, values = (var_988, x1_1))[name = string("op_991")]; + tensor sin_5 = transpose(perm = var_965, x = sin_1_cast_fp16)[name = string("transpose_123")]; + tensor var_992 = mul(x = var_991, y = sin_5)[name = string("op_992")]; + tensor query_states_3 = add(x = var_967, y = var_992)[name = string("query_states_3")]; + tensor var_995 = mul(x = k_1, y = cos_5)[name = string("op_995")]; + tensor x1_3_begin_0 = const()[name = string("x1_3_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_3_end_0 = const()[name = string("x1_3_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_3_end_mask_0 = const()[name = string("x1_3_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_3 = slice_by_index(begin = x1_3_begin_0, end = x1_3_end_0, end_mask = x1_3_end_mask_0, x = k_1)[name = string("x1_3")]; + tensor x2_3_begin_0 = const()[name = string("x2_3_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_3_end_0 = const()[name = string("x2_3_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_3_end_mask_0 = const()[name = string("x2_3_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_3 = slice_by_index(begin = x2_3_begin_0, end = x2_3_end_0, end_mask = x2_3_end_mask_0, x = k_1)[name = string("x2_3")]; + fp16 const_19_promoted = const()[name = string("const_19_promoted"), val = fp16(-0x1p+0)]; + tensor var_1016 = mul(x = x2_3, y = const_19_promoted)[name = string("op_1016")]; + int32 var_1018 = const()[name = string("op_1018"), val = int32(-1)]; + bool var_1019_interleave_0 = const()[name = string("op_1019_interleave_0"), val = bool(false)]; + tensor var_1019 = concat(axis = var_1018, interleave = var_1019_interleave_0, values = (var_1016, x1_3))[name = string("op_1019")]; + tensor var_1020 = mul(x = var_1019, y = sin_5)[name = string("op_1020")]; + tensor key_states_3 = add(x = var_995, y = var_1020)[name = string("key_states_3")]; + tensor seq_length_1 = const()[name = string("seq_length_1"), val = tensor([128])]; + tensor var_1042 = add(x = current_pos, y = seq_length_1)[name = string("op_1042")]; + tensor read_state_0 = read_state(input = model_model_kv_cache_0)[name = string("read_state_0")]; + tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([14])]; + tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; + tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; + tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([15])]; + int32 concat_2_axis_0 = const()[name = string("concat_2_axis_0"), val = int32(0)]; + bool concat_2_interleave_0 = const()[name = string("concat_2_interleave_0"), val = bool(false)]; + tensor concat_2 = concat(axis = concat_2_axis_0, interleave = concat_2_interleave_0, values = (expand_dims_0, expand_dims_1, current_pos, expand_dims_3))[name = string("concat_2")]; + tensor concat_3_values1_0 = const()[name = string("concat_3_values1_0"), val = tensor([0])]; + tensor concat_3_values3_0 = const()[name = string("concat_3_values3_0"), val = tensor([0])]; + int32 concat_3_axis_0 = const()[name = string("concat_3_axis_0"), val = int32(0)]; + bool concat_3_interleave_0 = const()[name = string("concat_3_interleave_0"), val = bool(false)]; + tensor concat_3 = concat(axis = concat_3_axis_0, interleave = concat_3_interleave_0, values = (expand_dims_4, concat_3_values1_0, var_1042, concat_3_values3_0))[name = string("concat_3")]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_2, begin_mask = model_model_kv_cache_0_internal_tensor_assign_1_begin_mask_0, end = concat_3, end_mask = model_model_kv_cache_0_internal_tensor_assign_1_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_1_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_1_stride_0, update = key_states_3, x = read_state_0)[name = string("model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_1_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_84_write_state")]; + tensor coreml_update_state_28 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_84")]; + tensor expand_dims_6 = const()[name = string("expand_dims_6"), val = tensor([42])]; + tensor expand_dims_7 = const()[name = string("expand_dims_7"), val = tensor([0])]; + tensor expand_dims_9 = const()[name = string("expand_dims_9"), val = tensor([0])]; + tensor expand_dims_10 = const()[name = string("expand_dims_10"), val = tensor([43])]; + int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; + bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; + tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_6, expand_dims_7, current_pos, expand_dims_9))[name = string("concat_6")]; + tensor concat_7_values1_0 = const()[name = string("concat_7_values1_0"), val = tensor([0])]; + tensor concat_7_values3_0 = const()[name = string("concat_7_values3_0"), val = tensor([0])]; + int32 concat_7_axis_0 = const()[name = string("concat_7_axis_0"), val = int32(0)]; + bool concat_7_interleave_0 = const()[name = string("concat_7_interleave_0"), val = bool(false)]; + tensor concat_7 = concat(axis = concat_7_axis_0, interleave = concat_7_interleave_0, values = (expand_dims_10, concat_7_values1_0, var_1042, concat_7_values3_0))[name = string("concat_7")]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_3 = transpose(perm = var_903, x = var_898)[name = string("transpose_122")]; + tensor model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_6, begin_mask = model_model_kv_cache_0_internal_tensor_assign_2_begin_mask_0, end = concat_7, end_mask = model_model_kv_cache_0_internal_tensor_assign_2_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_2_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_2_stride_0, update = value_states_3, x = coreml_update_state_28)[name = string("model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_2_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_85_write_state")]; + tensor coreml_update_state_29 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_85")]; + tensor var_1091_begin_0 = const()[name = string("op_1091_begin_0"), val = tensor([14, 0, 0, 0])]; + tensor var_1091_end_0 = const()[name = string("op_1091_end_0"), val = tensor([15, 8, 1024, 128])]; + tensor var_1091_end_mask_0 = const()[name = string("op_1091_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1091_cast_fp16 = slice_by_index(begin = var_1091_begin_0, end = var_1091_end_0, end_mask = var_1091_end_mask_0, x = coreml_update_state_29)[name = string("op_1091_cast_fp16")]; + tensor K_layer_cache_1_axes_0 = const()[name = string("K_layer_cache_1_axes_0"), val = tensor([0])]; + tensor K_layer_cache_1_cast_fp16 = squeeze(axes = K_layer_cache_1_axes_0, x = var_1091_cast_fp16)[name = string("K_layer_cache_1_cast_fp16")]; + tensor var_1098_begin_0 = const()[name = string("op_1098_begin_0"), val = tensor([42, 0, 0, 0])]; + tensor var_1098_end_0 = const()[name = string("op_1098_end_0"), val = tensor([43, 8, 1024, 128])]; + tensor var_1098_end_mask_0 = const()[name = string("op_1098_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1098_cast_fp16 = slice_by_index(begin = var_1098_begin_0, end = var_1098_end_0, end_mask = var_1098_end_mask_0, x = coreml_update_state_29)[name = string("op_1098_cast_fp16")]; + tensor V_layer_cache_1_axes_0 = const()[name = string("V_layer_cache_1_axes_0"), val = tensor([0])]; + tensor V_layer_cache_1_cast_fp16 = squeeze(axes = V_layer_cache_1_axes_0, x = var_1098_cast_fp16)[name = string("V_layer_cache_1_cast_fp16")]; + tensor x_3_axes_0 = const()[name = string("x_3_axes_0"), val = tensor([1])]; + tensor x_3_cast_fp16 = expand_dims(axes = x_3_axes_0, x = K_layer_cache_1_cast_fp16)[name = string("x_3_cast_fp16")]; + tensor var_1127 = const()[name = string("op_1127"), val = tensor([1, 2, 1, 1])]; + tensor x_5_cast_fp16 = tile(reps = var_1127, x = x_3_cast_fp16)[name = string("x_5_cast_fp16")]; + tensor var_1139 = const()[name = string("op_1139"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_7_cast_fp16 = reshape(shape = var_1139, x = x_5_cast_fp16)[name = string("key_states_7_cast_fp16")]; + tensor x_9_axes_0 = const()[name = string("x_9_axes_0"), val = tensor([1])]; + tensor x_9_cast_fp16 = expand_dims(axes = x_9_axes_0, x = V_layer_cache_1_cast_fp16)[name = string("x_9_cast_fp16")]; + tensor var_1147 = const()[name = string("op_1147"), val = tensor([1, 2, 1, 1])]; + tensor x_11_cast_fp16 = tile(reps = var_1147, x = x_9_cast_fp16)[name = string("x_11_cast_fp16")]; + bool var_1174_transpose_x_0 = const()[name = string("op_1174_transpose_x_0"), val = bool(false)]; + bool var_1174_transpose_y_0 = const()[name = string("op_1174_transpose_y_0"), val = bool(true)]; + tensor var_1174 = matmul(transpose_x = var_1174_transpose_x_0, transpose_y = var_1174_transpose_y_0, x = query_states_3, y = key_states_7_cast_fp16)[name = string("op_1174")]; + fp16 var_1175_to_fp16 = const()[name = string("op_1175_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_1_cast_fp16 = mul(x = var_1174, y = var_1175_to_fp16)[name = string("attn_weights_1_cast_fp16")]; + tensor attn_weights_3_cast_fp16 = add(x = attn_weights_1_cast_fp16, y = causal_mask)[name = string("attn_weights_3_cast_fp16")]; + int32 var_1210 = const()[name = string("op_1210"), val = int32(-1)]; + tensor var_1212_cast_fp16 = softmax(axis = var_1210, x = attn_weights_3_cast_fp16)[name = string("op_1212_cast_fp16")]; + tensor concat_12 = const()[name = string("concat_12"), val = tensor([16, 128, 1024])]; + tensor reshape_0_cast_fp16 = reshape(shape = concat_12, x = var_1212_cast_fp16)[name = string("reshape_0_cast_fp16")]; + tensor concat_13 = const()[name = string("concat_13"), val = tensor([16, 1024, 128])]; + tensor reshape_1_cast_fp16 = reshape(shape = concat_13, x = x_11_cast_fp16)[name = string("reshape_1_cast_fp16")]; + bool matmul_0_transpose_x_0 = const()[name = string("matmul_0_transpose_x_0"), val = bool(false)]; + bool matmul_0_transpose_y_0 = const()[name = string("matmul_0_transpose_y_0"), val = bool(false)]; + tensor matmul_0_cast_fp16 = matmul(transpose_x = matmul_0_transpose_x_0, transpose_y = matmul_0_transpose_y_0, x = reshape_0_cast_fp16, y = reshape_1_cast_fp16)[name = string("matmul_0_cast_fp16")]; + tensor concat_17 = const()[name = string("concat_17"), val = tensor([1, 16, 128, 128])]; + tensor reshape_2_cast_fp16 = reshape(shape = concat_17, x = matmul_0_cast_fp16)[name = string("reshape_2_cast_fp16")]; + tensor var_1224_perm_0 = const()[name = string("op_1224_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1243 = const()[name = string("op_1243"), val = tensor([1, 128, 2048])]; + tensor var_1224_cast_fp16 = transpose(perm = var_1224_perm_0, x = reshape_2_cast_fp16)[name = string("transpose_121")]; + tensor attn_output_5_cast_fp16 = reshape(shape = var_1243, x = var_1224_cast_fp16)[name = string("attn_output_5_cast_fp16")]; + tensor var_1248 = const()[name = string("op_1248"), val = tensor([0, 2, 1])]; + string var_1264_pad_type_0 = const()[name = string("op_1264_pad_type_0"), val = string("valid")]; + int32 var_1264_groups_0 = const()[name = string("op_1264_groups_0"), val = int32(1)]; + tensor var_1264_strides_0 = const()[name = string("op_1264_strides_0"), val = tensor([1])]; + tensor var_1264_pad_0 = const()[name = string("op_1264_pad_0"), val = tensor([0, 0])]; + tensor var_1264_dilations_0 = const()[name = string("op_1264_dilations_0"), val = tensor([1])]; + tensor squeeze_0_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(663502208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667696576))))[name = string("squeeze_0_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_1249_cast_fp16 = transpose(perm = var_1248, x = attn_output_5_cast_fp16)[name = string("transpose_120")]; + tensor var_1264_cast_fp16 = conv(dilations = var_1264_dilations_0, groups = var_1264_groups_0, pad = var_1264_pad_0, pad_type = var_1264_pad_type_0, strides = var_1264_strides_0, weight = squeeze_0_cast_fp16_to_fp32_to_fp16_palettized, x = var_1249_cast_fp16)[name = string("op_1264_cast_fp16")]; + tensor var_1268 = const()[name = string("op_1268"), val = tensor([0, 2, 1])]; + tensor attn_output_9_cast_fp16 = transpose(perm = var_1268, x = var_1264_cast_fp16)[name = string("transpose_119")]; + tensor hidden_states_9_cast_fp16 = add(x = hidden_states, y = attn_output_9_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; + int32 var_1281 = const()[name = string("op_1281"), val = int32(-1)]; + fp16 const_31_promoted_to_fp16 = const()[name = string("const_31_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1283_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_31_promoted_to_fp16)[name = string("op_1283_cast_fp16")]; + bool input_11_interleave_0 = const()[name = string("input_11_interleave_0"), val = bool(false)]; + tensor input_11_cast_fp16 = concat(axis = var_1281, interleave = input_11_interleave_0, values = (hidden_states_9_cast_fp16, var_1283_cast_fp16))[name = string("input_11_cast_fp16")]; + tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; + fp16 var_1278_to_fp16 = const()[name = string("op_1278_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_1278_to_fp16, x = input_11_cast_fp16)[name = string("normed_13_cast_fp16")]; + tensor normed_15_begin_0 = const()[name = string("normed_15_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_15_end_0 = const()[name = string("normed_15_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_15_end_mask_0 = const()[name = string("normed_15_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_15_cast_fp16 = slice_by_index(begin = normed_15_begin_0, end = normed_15_end_0, end_mask = normed_15_end_mask_0, x = normed_13_cast_fp16)[name = string("normed_15_cast_fp16")]; + tensor const_34_promoted_to_fp16 = const()[name = string("const_34_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667827712)))]; + tensor x_13_cast_fp16 = mul(x = normed_15_cast_fp16, y = const_34_promoted_to_fp16)[name = string("x_13_cast_fp16")]; + tensor var_1308 = const()[name = string("op_1308"), val = tensor([0, 2, 1])]; + tensor input_13_axes_0 = const()[name = string("input_13_axes_0"), val = tensor([2])]; + tensor var_1309 = transpose(perm = var_1308, x = x_13_cast_fp16)[name = string("transpose_118")]; + tensor input_13 = expand_dims(axes = input_13_axes_0, x = var_1309)[name = string("input_13")]; + string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; + tensor input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor([1, 1])]; + tensor input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor([1, 1])]; + int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; + tensor input_15 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = model_model_layers_14_mlp_gate_proj_weight_palettized, x = input_13)[name = string("input_15")]; + string b_1_pad_type_0 = const()[name = string("b_1_pad_type_0"), val = string("valid")]; + tensor b_1_strides_0 = const()[name = string("b_1_strides_0"), val = tensor([1, 1])]; + tensor b_1_pad_0 = const()[name = string("b_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_1_dilations_0 = const()[name = string("b_1_dilations_0"), val = tensor([1, 1])]; + int32 b_1_groups_0 = const()[name = string("b_1_groups_0"), val = int32(1)]; + tensor b_1 = conv(dilations = b_1_dilations_0, groups = b_1_groups_0, pad = b_1_pad_0, pad_type = b_1_pad_type_0, strides = b_1_strides_0, weight = model_model_layers_14_mlp_up_proj_weight_palettized, x = input_13)[name = string("b_1")]; + tensor c_1 = silu(x = input_15)[name = string("c_1")]; + tensor input_17 = mul(x = c_1, y = b_1)[name = string("input_17")]; + string e_1_pad_type_0 = const()[name = string("e_1_pad_type_0"), val = string("valid")]; + tensor e_1_strides_0 = const()[name = string("e_1_strides_0"), val = tensor([1, 1])]; + tensor e_1_pad_0 = const()[name = string("e_1_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_1_dilations_0 = const()[name = string("e_1_dilations_0"), val = tensor([1, 1])]; + int32 e_1_groups_0 = const()[name = string("e_1_groups_0"), val = int32(1)]; + tensor e_1 = conv(dilations = e_1_dilations_0, groups = e_1_groups_0, pad = e_1_pad_0, pad_type = e_1_pad_type_0, strides = e_1_strides_0, weight = model_model_layers_14_mlp_down_proj_weight_palettized, x = input_17)[name = string("e_1")]; + tensor var_1331_axes_0 = const()[name = string("op_1331_axes_0"), val = tensor([2])]; + tensor var_1331 = squeeze(axes = var_1331_axes_0, x = e_1)[name = string("op_1331")]; + tensor var_1332 = const()[name = string("op_1332"), val = tensor([0, 2, 1])]; + tensor var_1333 = transpose(perm = var_1332, x = var_1331)[name = string("transpose_117")]; + tensor hidden_states_11_cast_fp16 = add(x = hidden_states_9_cast_fp16, y = var_1333)[name = string("hidden_states_11_cast_fp16")]; + int32 var_1345 = const()[name = string("op_1345"), val = int32(-1)]; + fp16 const_35_promoted_to_fp16 = const()[name = string("const_35_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1347_cast_fp16 = mul(x = hidden_states_11_cast_fp16, y = const_35_promoted_to_fp16)[name = string("op_1347_cast_fp16")]; + bool input_19_interleave_0 = const()[name = string("input_19_interleave_0"), val = bool(false)]; + tensor input_19_cast_fp16 = concat(axis = var_1345, interleave = input_19_interleave_0, values = (hidden_states_11_cast_fp16, var_1347_cast_fp16))[name = string("input_19_cast_fp16")]; + tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; + fp16 var_1342_to_fp16 = const()[name = string("op_1342_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_1342_to_fp16, x = input_19_cast_fp16)[name = string("normed_17_cast_fp16")]; + tensor normed_19_begin_0 = const()[name = string("normed_19_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_19_end_0 = const()[name = string("normed_19_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_19_end_mask_0 = const()[name = string("normed_19_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_19_cast_fp16 = slice_by_index(begin = normed_19_begin_0, end = normed_19_end_0, end_mask = normed_19_end_mask_0, x = normed_17_cast_fp16)[name = string("normed_19_cast_fp16")]; + tensor const_38_promoted_to_fp16 = const()[name = string("const_38_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667831872)))]; + tensor hidden_states_13_cast_fp16 = mul(x = normed_19_cast_fp16, y = const_38_promoted_to_fp16)[name = string("hidden_states_13_cast_fp16")]; + tensor var_1370 = const()[name = string("op_1370"), val = tensor([0, 2, 1])]; + tensor var_1373_axes_0 = const()[name = string("op_1373_axes_0"), val = tensor([2])]; + tensor var_1371_cast_fp16 = transpose(perm = var_1370, x = hidden_states_13_cast_fp16)[name = string("transpose_116")]; + tensor var_1373_cast_fp16 = expand_dims(axes = var_1373_axes_0, x = var_1371_cast_fp16)[name = string("op_1373_cast_fp16")]; + string query_states_9_pad_type_0 = const()[name = string("query_states_9_pad_type_0"), val = string("valid")]; + tensor query_states_9_strides_0 = const()[name = string("query_states_9_strides_0"), val = tensor([1, 1])]; + tensor query_states_9_pad_0 = const()[name = string("query_states_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_9_dilations_0 = const()[name = string("query_states_9_dilations_0"), val = tensor([1, 1])]; + int32 query_states_9_groups_0 = const()[name = string("query_states_9_groups_0"), val = int32(1)]; + tensor query_states_9 = conv(dilations = query_states_9_dilations_0, groups = query_states_9_groups_0, pad = query_states_9_pad_0, pad_type = query_states_9_pad_type_0, strides = query_states_9_strides_0, weight = model_model_layers_15_self_attn_q_proj_weight_palettized, x = var_1373_cast_fp16)[name = string("query_states_9")]; + string key_states_11_pad_type_0 = const()[name = string("key_states_11_pad_type_0"), val = string("valid")]; + tensor key_states_11_strides_0 = const()[name = string("key_states_11_strides_0"), val = tensor([1, 1])]; + tensor key_states_11_pad_0 = const()[name = string("key_states_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_11_dilations_0 = const()[name = string("key_states_11_dilations_0"), val = tensor([1, 1])]; + int32 key_states_11_groups_0 = const()[name = string("key_states_11_groups_0"), val = int32(1)]; + tensor key_states_11 = conv(dilations = key_states_11_dilations_0, groups = key_states_11_groups_0, pad = key_states_11_pad_0, pad_type = key_states_11_pad_type_0, strides = key_states_11_strides_0, weight = model_model_layers_15_self_attn_k_proj_weight_palettized, x = var_1373_cast_fp16)[name = string("key_states_11")]; + string value_states_9_pad_type_0 = const()[name = string("value_states_9_pad_type_0"), val = string("valid")]; + tensor value_states_9_strides_0 = const()[name = string("value_states_9_strides_0"), val = tensor([1, 1])]; + tensor value_states_9_pad_0 = const()[name = string("value_states_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_9_dilations_0 = const()[name = string("value_states_9_dilations_0"), val = tensor([1, 1])]; + int32 value_states_9_groups_0 = const()[name = string("value_states_9_groups_0"), val = int32(1)]; + tensor value_states_9 = conv(dilations = value_states_9_dilations_0, groups = value_states_9_groups_0, pad = value_states_9_pad_0, pad_type = value_states_9_pad_type_0, strides = value_states_9_strides_0, weight = model_model_layers_15_self_attn_v_proj_weight_palettized, x = var_1373_cast_fp16)[name = string("value_states_9")]; + tensor var_1415 = const()[name = string("op_1415"), val = tensor([1, 16, 128, 128])]; + tensor var_1416 = reshape(shape = var_1415, x = query_states_9)[name = string("op_1416")]; + tensor var_1421 = const()[name = string("op_1421"), val = tensor([0, 1, 3, 2])]; + tensor var_1426 = const()[name = string("op_1426"), val = tensor([1, 8, 128, 128])]; + tensor var_1427 = reshape(shape = var_1426, x = key_states_11)[name = string("op_1427")]; + tensor var_1432 = const()[name = string("op_1432"), val = tensor([0, 1, 3, 2])]; + tensor var_1437 = const()[name = string("op_1437"), val = tensor([1, 8, 128, 128])]; + tensor var_1438 = reshape(shape = var_1437, x = value_states_9)[name = string("op_1438")]; + tensor var_1443 = const()[name = string("op_1443"), val = tensor([0, 1, 3, 2])]; + int32 var_1454 = const()[name = string("op_1454"), val = int32(-1)]; + fp16 const_40_promoted = const()[name = string("const_40_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_15 = transpose(perm = var_1421, x = var_1416)[name = string("transpose_115")]; + tensor var_1456 = mul(x = hidden_states_15, y = const_40_promoted)[name = string("op_1456")]; + bool input_23_interleave_0 = const()[name = string("input_23_interleave_0"), val = bool(false)]; + tensor input_23 = concat(axis = var_1454, interleave = input_23_interleave_0, values = (hidden_states_15, var_1456))[name = string("input_23")]; + tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; + fp16 var_1451_to_fp16 = const()[name = string("op_1451_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_1451_to_fp16, x = input_23)[name = string("normed_21_cast_fp16")]; + tensor normed_23_begin_0 = const()[name = string("normed_23_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_23_end_0 = const()[name = string("normed_23_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_23_end_mask_0 = const()[name = string("normed_23_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_23 = slice_by_index(begin = normed_23_begin_0, end = normed_23_end_0, end_mask = normed_23_end_mask_0, x = normed_21_cast_fp16)[name = string("normed_23")]; + tensor const_43 = const()[name = string("const_43"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667836032)))]; + tensor q_3 = mul(x = normed_23, y = const_43)[name = string("q_3")]; + int32 var_1479 = const()[name = string("op_1479"), val = int32(-1)]; + fp16 const_44_promoted = const()[name = string("const_44_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_17 = transpose(perm = var_1432, x = var_1427)[name = string("transpose_114")]; + tensor var_1481 = mul(x = hidden_states_17, y = const_44_promoted)[name = string("op_1481")]; + bool input_25_interleave_0 = const()[name = string("input_25_interleave_0"), val = bool(false)]; + tensor input_25 = concat(axis = var_1479, interleave = input_25_interleave_0, values = (hidden_states_17, var_1481))[name = string("input_25")]; + tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; + fp16 var_1476_to_fp16 = const()[name = string("op_1476_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_1476_to_fp16, x = input_25)[name = string("normed_25_cast_fp16")]; + tensor normed_27_begin_0 = const()[name = string("normed_27_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_27_end_0 = const()[name = string("normed_27_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_27_end_mask_0 = const()[name = string("normed_27_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_27 = slice_by_index(begin = normed_27_begin_0, end = normed_27_end_0, end_mask = normed_27_end_mask_0, x = normed_25_cast_fp16)[name = string("normed_27")]; + tensor const_47 = const()[name = string("const_47"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667836352)))]; + tensor k_3 = mul(x = normed_27, y = const_47)[name = string("k_3")]; + tensor var_1507 = mul(x = q_3, y = cos_5)[name = string("op_1507")]; + tensor x1_5_begin_0 = const()[name = string("x1_5_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_5_end_0 = const()[name = string("x1_5_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_5_end_mask_0 = const()[name = string("x1_5_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_5 = slice_by_index(begin = x1_5_begin_0, end = x1_5_end_0, end_mask = x1_5_end_mask_0, x = q_3)[name = string("x1_5")]; + tensor x2_5_begin_0 = const()[name = string("x2_5_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_5_end_0 = const()[name = string("x2_5_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_5_end_mask_0 = const()[name = string("x2_5_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_5 = slice_by_index(begin = x2_5_begin_0, end = x2_5_end_0, end_mask = x2_5_end_mask_0, x = q_3)[name = string("x2_5")]; + fp16 const_50_promoted = const()[name = string("const_50_promoted"), val = fp16(-0x1p+0)]; + tensor var_1528 = mul(x = x2_5, y = const_50_promoted)[name = string("op_1528")]; + int32 var_1530 = const()[name = string("op_1530"), val = int32(-1)]; + bool var_1531_interleave_0 = const()[name = string("op_1531_interleave_0"), val = bool(false)]; + tensor var_1531 = concat(axis = var_1530, interleave = var_1531_interleave_0, values = (var_1528, x1_5))[name = string("op_1531")]; + tensor var_1532 = mul(x = var_1531, y = sin_5)[name = string("op_1532")]; + tensor query_states_11 = add(x = var_1507, y = var_1532)[name = string("query_states_11")]; + tensor var_1535 = mul(x = k_3, y = cos_5)[name = string("op_1535")]; + tensor x1_7_begin_0 = const()[name = string("x1_7_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_7_end_0 = const()[name = string("x1_7_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_7_end_mask_0 = const()[name = string("x1_7_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_7 = slice_by_index(begin = x1_7_begin_0, end = x1_7_end_0, end_mask = x1_7_end_mask_0, x = k_3)[name = string("x1_7")]; + tensor x2_7_begin_0 = const()[name = string("x2_7_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_7_end_0 = const()[name = string("x2_7_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_7_end_mask_0 = const()[name = string("x2_7_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_7 = slice_by_index(begin = x2_7_begin_0, end = x2_7_end_0, end_mask = x2_7_end_mask_0, x = k_3)[name = string("x2_7")]; + fp16 const_53_promoted = const()[name = string("const_53_promoted"), val = fp16(-0x1p+0)]; + tensor var_1556 = mul(x = x2_7, y = const_53_promoted)[name = string("op_1556")]; + int32 var_1558 = const()[name = string("op_1558"), val = int32(-1)]; + bool var_1559_interleave_0 = const()[name = string("op_1559_interleave_0"), val = bool(false)]; + tensor var_1559 = concat(axis = var_1558, interleave = var_1559_interleave_0, values = (var_1556, x1_7))[name = string("op_1559")]; + tensor var_1560 = mul(x = var_1559, y = sin_5)[name = string("op_1560")]; + tensor key_states_13 = add(x = var_1535, y = var_1560)[name = string("key_states_13")]; + tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([15])]; + tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; + tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; + tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([16])]; + int32 concat_20_axis_0 = const()[name = string("concat_20_axis_0"), val = int32(0)]; + bool concat_20_interleave_0 = const()[name = string("concat_20_interleave_0"), val = bool(false)]; + tensor concat_20 = concat(axis = concat_20_axis_0, interleave = concat_20_interleave_0, values = (expand_dims_12, expand_dims_13, current_pos, expand_dims_15))[name = string("concat_20")]; + tensor concat_21_values1_0 = const()[name = string("concat_21_values1_0"), val = tensor([0])]; + tensor concat_21_values3_0 = const()[name = string("concat_21_values3_0"), val = tensor([0])]; + int32 concat_21_axis_0 = const()[name = string("concat_21_axis_0"), val = int32(0)]; + bool concat_21_interleave_0 = const()[name = string("concat_21_interleave_0"), val = bool(false)]; + tensor concat_21 = concat(axis = concat_21_axis_0, interleave = concat_21_interleave_0, values = (expand_dims_16, concat_21_values1_0, var_1042, concat_21_values3_0))[name = string("concat_21")]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_20, begin_mask = model_model_kv_cache_0_internal_tensor_assign_3_begin_mask_0, end = concat_21, end_mask = model_model_kv_cache_0_internal_tensor_assign_3_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_3_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_3_stride_0, update = key_states_13, x = coreml_update_state_29)[name = string("model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_3_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_86_write_state")]; + tensor coreml_update_state_30 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_86")]; + tensor expand_dims_18 = const()[name = string("expand_dims_18"), val = tensor([43])]; + tensor expand_dims_19 = const()[name = string("expand_dims_19"), val = tensor([0])]; + tensor expand_dims_21 = const()[name = string("expand_dims_21"), val = tensor([0])]; + tensor expand_dims_22 = const()[name = string("expand_dims_22"), val = tensor([44])]; + int32 concat_24_axis_0 = const()[name = string("concat_24_axis_0"), val = int32(0)]; + bool concat_24_interleave_0 = const()[name = string("concat_24_interleave_0"), val = bool(false)]; + tensor concat_24 = concat(axis = concat_24_axis_0, interleave = concat_24_interleave_0, values = (expand_dims_18, expand_dims_19, current_pos, expand_dims_21))[name = string("concat_24")]; + tensor concat_25_values1_0 = const()[name = string("concat_25_values1_0"), val = tensor([0])]; + tensor concat_25_values3_0 = const()[name = string("concat_25_values3_0"), val = tensor([0])]; + int32 concat_25_axis_0 = const()[name = string("concat_25_axis_0"), val = int32(0)]; + bool concat_25_interleave_0 = const()[name = string("concat_25_interleave_0"), val = bool(false)]; + tensor concat_25 = concat(axis = concat_25_axis_0, interleave = concat_25_interleave_0, values = (expand_dims_22, concat_25_values1_0, var_1042, concat_25_values3_0))[name = string("concat_25")]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_11 = transpose(perm = var_1443, x = var_1438)[name = string("transpose_113")]; + tensor model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_24, begin_mask = model_model_kv_cache_0_internal_tensor_assign_4_begin_mask_0, end = concat_25, end_mask = model_model_kv_cache_0_internal_tensor_assign_4_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_4_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_4_stride_0, update = value_states_11, x = coreml_update_state_30)[name = string("model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_4_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_87_write_state")]; + tensor coreml_update_state_31 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_87")]; + tensor var_1631_begin_0 = const()[name = string("op_1631_begin_0"), val = tensor([15, 0, 0, 0])]; + tensor var_1631_end_0 = const()[name = string("op_1631_end_0"), val = tensor([16, 8, 1024, 128])]; + tensor var_1631_end_mask_0 = const()[name = string("op_1631_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1631_cast_fp16 = slice_by_index(begin = var_1631_begin_0, end = var_1631_end_0, end_mask = var_1631_end_mask_0, x = coreml_update_state_31)[name = string("op_1631_cast_fp16")]; + tensor K_layer_cache_3_axes_0 = const()[name = string("K_layer_cache_3_axes_0"), val = tensor([0])]; + tensor K_layer_cache_3_cast_fp16 = squeeze(axes = K_layer_cache_3_axes_0, x = var_1631_cast_fp16)[name = string("K_layer_cache_3_cast_fp16")]; + tensor var_1638_begin_0 = const()[name = string("op_1638_begin_0"), val = tensor([43, 0, 0, 0])]; + tensor var_1638_end_0 = const()[name = string("op_1638_end_0"), val = tensor([44, 8, 1024, 128])]; + tensor var_1638_end_mask_0 = const()[name = string("op_1638_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_1638_cast_fp16 = slice_by_index(begin = var_1638_begin_0, end = var_1638_end_0, end_mask = var_1638_end_mask_0, x = coreml_update_state_31)[name = string("op_1638_cast_fp16")]; + tensor V_layer_cache_3_axes_0 = const()[name = string("V_layer_cache_3_axes_0"), val = tensor([0])]; + tensor V_layer_cache_3_cast_fp16 = squeeze(axes = V_layer_cache_3_axes_0, x = var_1638_cast_fp16)[name = string("V_layer_cache_3_cast_fp16")]; + tensor x_19_axes_0 = const()[name = string("x_19_axes_0"), val = tensor([1])]; + tensor x_19_cast_fp16 = expand_dims(axes = x_19_axes_0, x = K_layer_cache_3_cast_fp16)[name = string("x_19_cast_fp16")]; + tensor var_1667 = const()[name = string("op_1667"), val = tensor([1, 2, 1, 1])]; + tensor x_21_cast_fp16 = tile(reps = var_1667, x = x_19_cast_fp16)[name = string("x_21_cast_fp16")]; + tensor var_1679 = const()[name = string("op_1679"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_17_cast_fp16 = reshape(shape = var_1679, x = x_21_cast_fp16)[name = string("key_states_17_cast_fp16")]; + tensor x_25_axes_0 = const()[name = string("x_25_axes_0"), val = tensor([1])]; + tensor x_25_cast_fp16 = expand_dims(axes = x_25_axes_0, x = V_layer_cache_3_cast_fp16)[name = string("x_25_cast_fp16")]; + tensor var_1687 = const()[name = string("op_1687"), val = tensor([1, 2, 1, 1])]; + tensor x_27_cast_fp16 = tile(reps = var_1687, x = x_25_cast_fp16)[name = string("x_27_cast_fp16")]; + bool var_1714_transpose_x_0 = const()[name = string("op_1714_transpose_x_0"), val = bool(false)]; + bool var_1714_transpose_y_0 = const()[name = string("op_1714_transpose_y_0"), val = bool(true)]; + tensor var_1714 = matmul(transpose_x = var_1714_transpose_x_0, transpose_y = var_1714_transpose_y_0, x = query_states_11, y = key_states_17_cast_fp16)[name = string("op_1714")]; + fp16 var_1715_to_fp16 = const()[name = string("op_1715_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_5_cast_fp16 = mul(x = var_1714, y = var_1715_to_fp16)[name = string("attn_weights_5_cast_fp16")]; + tensor attn_weights_7_cast_fp16 = add(x = attn_weights_5_cast_fp16, y = causal_mask)[name = string("attn_weights_7_cast_fp16")]; + int32 var_1750 = const()[name = string("op_1750"), val = int32(-1)]; + tensor var_1752_cast_fp16 = softmax(axis = var_1750, x = attn_weights_7_cast_fp16)[name = string("op_1752_cast_fp16")]; + tensor concat_30 = const()[name = string("concat_30"), val = tensor([16, 128, 1024])]; + tensor reshape_3_cast_fp16 = reshape(shape = concat_30, x = var_1752_cast_fp16)[name = string("reshape_3_cast_fp16")]; + tensor concat_31 = const()[name = string("concat_31"), val = tensor([16, 1024, 128])]; + tensor reshape_4_cast_fp16 = reshape(shape = concat_31, x = x_27_cast_fp16)[name = string("reshape_4_cast_fp16")]; + bool matmul_1_transpose_x_0 = const()[name = string("matmul_1_transpose_x_0"), val = bool(false)]; + bool matmul_1_transpose_y_0 = const()[name = string("matmul_1_transpose_y_0"), val = bool(false)]; + tensor matmul_1_cast_fp16 = matmul(transpose_x = matmul_1_transpose_x_0, transpose_y = matmul_1_transpose_y_0, x = reshape_3_cast_fp16, y = reshape_4_cast_fp16)[name = string("matmul_1_cast_fp16")]; + tensor concat_35 = const()[name = string("concat_35"), val = tensor([1, 16, 128, 128])]; + tensor reshape_5_cast_fp16 = reshape(shape = concat_35, x = matmul_1_cast_fp16)[name = string("reshape_5_cast_fp16")]; + tensor var_1764_perm_0 = const()[name = string("op_1764_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_1783 = const()[name = string("op_1783"), val = tensor([1, 128, 2048])]; + tensor var_1764_cast_fp16 = transpose(perm = var_1764_perm_0, x = reshape_5_cast_fp16)[name = string("transpose_112")]; + tensor attn_output_15_cast_fp16 = reshape(shape = var_1783, x = var_1764_cast_fp16)[name = string("attn_output_15_cast_fp16")]; + tensor var_1788 = const()[name = string("op_1788"), val = tensor([0, 2, 1])]; + string var_1804_pad_type_0 = const()[name = string("op_1804_pad_type_0"), val = string("valid")]; + int32 var_1804_groups_0 = const()[name = string("op_1804_groups_0"), val = int32(1)]; + tensor var_1804_strides_0 = const()[name = string("op_1804_strides_0"), val = tensor([1])]; + tensor var_1804_pad_0 = const()[name = string("op_1804_pad_0"), val = tensor([0, 0])]; + tensor var_1804_dilations_0 = const()[name = string("op_1804_dilations_0"), val = tensor([1])]; + tensor squeeze_1_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(667836672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672031040))))[name = string("squeeze_1_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_1789_cast_fp16 = transpose(perm = var_1788, x = attn_output_15_cast_fp16)[name = string("transpose_111")]; + tensor var_1804_cast_fp16 = conv(dilations = var_1804_dilations_0, groups = var_1804_groups_0, pad = var_1804_pad_0, pad_type = var_1804_pad_type_0, strides = var_1804_strides_0, weight = squeeze_1_cast_fp16_to_fp32_to_fp16_palettized, x = var_1789_cast_fp16)[name = string("op_1804_cast_fp16")]; + tensor var_1808 = const()[name = string("op_1808"), val = tensor([0, 2, 1])]; + tensor attn_output_19_cast_fp16 = transpose(perm = var_1808, x = var_1804_cast_fp16)[name = string("transpose_110")]; + tensor hidden_states_19_cast_fp16 = add(x = hidden_states_11_cast_fp16, y = attn_output_19_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; + int32 var_1821 = const()[name = string("op_1821"), val = int32(-1)]; + fp16 const_65_promoted_to_fp16 = const()[name = string("const_65_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1823_cast_fp16 = mul(x = hidden_states_19_cast_fp16, y = const_65_promoted_to_fp16)[name = string("op_1823_cast_fp16")]; + bool input_29_interleave_0 = const()[name = string("input_29_interleave_0"), val = bool(false)]; + tensor input_29_cast_fp16 = concat(axis = var_1821, interleave = input_29_interleave_0, values = (hidden_states_19_cast_fp16, var_1823_cast_fp16))[name = string("input_29_cast_fp16")]; + tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; + fp16 var_1818_to_fp16 = const()[name = string("op_1818_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_1818_to_fp16, x = input_29_cast_fp16)[name = string("normed_29_cast_fp16")]; + tensor normed_31_begin_0 = const()[name = string("normed_31_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_31_end_0 = const()[name = string("normed_31_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_31_end_mask_0 = const()[name = string("normed_31_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_31_cast_fp16 = slice_by_index(begin = normed_31_begin_0, end = normed_31_end_0, end_mask = normed_31_end_mask_0, x = normed_29_cast_fp16)[name = string("normed_31_cast_fp16")]; + tensor const_68_promoted_to_fp16 = const()[name = string("const_68_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672162176)))]; + tensor x_29_cast_fp16 = mul(x = normed_31_cast_fp16, y = const_68_promoted_to_fp16)[name = string("x_29_cast_fp16")]; + tensor var_1848 = const()[name = string("op_1848"), val = tensor([0, 2, 1])]; + tensor input_31_axes_0 = const()[name = string("input_31_axes_0"), val = tensor([2])]; + tensor var_1849 = transpose(perm = var_1848, x = x_29_cast_fp16)[name = string("transpose_109")]; + tensor input_31 = expand_dims(axes = input_31_axes_0, x = var_1849)[name = string("input_31")]; + string input_33_pad_type_0 = const()[name = string("input_33_pad_type_0"), val = string("valid")]; + tensor input_33_strides_0 = const()[name = string("input_33_strides_0"), val = tensor([1, 1])]; + tensor input_33_pad_0 = const()[name = string("input_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_33_dilations_0 = const()[name = string("input_33_dilations_0"), val = tensor([1, 1])]; + int32 input_33_groups_0 = const()[name = string("input_33_groups_0"), val = int32(1)]; + tensor input_33 = conv(dilations = input_33_dilations_0, groups = input_33_groups_0, pad = input_33_pad_0, pad_type = input_33_pad_type_0, strides = input_33_strides_0, weight = model_model_layers_15_mlp_gate_proj_weight_palettized, x = input_31)[name = string("input_33")]; + string b_3_pad_type_0 = const()[name = string("b_3_pad_type_0"), val = string("valid")]; + tensor b_3_strides_0 = const()[name = string("b_3_strides_0"), val = tensor([1, 1])]; + tensor b_3_pad_0 = const()[name = string("b_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_3_dilations_0 = const()[name = string("b_3_dilations_0"), val = tensor([1, 1])]; + int32 b_3_groups_0 = const()[name = string("b_3_groups_0"), val = int32(1)]; + tensor b_3 = conv(dilations = b_3_dilations_0, groups = b_3_groups_0, pad = b_3_pad_0, pad_type = b_3_pad_type_0, strides = b_3_strides_0, weight = model_model_layers_15_mlp_up_proj_weight_palettized, x = input_31)[name = string("b_3")]; + tensor c_3 = silu(x = input_33)[name = string("c_3")]; + tensor input_35 = mul(x = c_3, y = b_3)[name = string("input_35")]; + string e_3_pad_type_0 = const()[name = string("e_3_pad_type_0"), val = string("valid")]; + tensor e_3_strides_0 = const()[name = string("e_3_strides_0"), val = tensor([1, 1])]; + tensor e_3_pad_0 = const()[name = string("e_3_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_3_dilations_0 = const()[name = string("e_3_dilations_0"), val = tensor([1, 1])]; + int32 e_3_groups_0 = const()[name = string("e_3_groups_0"), val = int32(1)]; + tensor e_3 = conv(dilations = e_3_dilations_0, groups = e_3_groups_0, pad = e_3_pad_0, pad_type = e_3_pad_type_0, strides = e_3_strides_0, weight = model_model_layers_15_mlp_down_proj_weight_palettized, x = input_35)[name = string("e_3")]; + tensor var_1871_axes_0 = const()[name = string("op_1871_axes_0"), val = tensor([2])]; + tensor var_1871 = squeeze(axes = var_1871_axes_0, x = e_3)[name = string("op_1871")]; + tensor var_1872 = const()[name = string("op_1872"), val = tensor([0, 2, 1])]; + tensor var_1873 = transpose(perm = var_1872, x = var_1871)[name = string("transpose_108")]; + tensor hidden_states_21_cast_fp16 = add(x = hidden_states_19_cast_fp16, y = var_1873)[name = string("hidden_states_21_cast_fp16")]; + int32 var_1885 = const()[name = string("op_1885"), val = int32(-1)]; + fp16 const_69_promoted_to_fp16 = const()[name = string("const_69_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_1887_cast_fp16 = mul(x = hidden_states_21_cast_fp16, y = const_69_promoted_to_fp16)[name = string("op_1887_cast_fp16")]; + bool input_37_interleave_0 = const()[name = string("input_37_interleave_0"), val = bool(false)]; + tensor input_37_cast_fp16 = concat(axis = var_1885, interleave = input_37_interleave_0, values = (hidden_states_21_cast_fp16, var_1887_cast_fp16))[name = string("input_37_cast_fp16")]; + tensor normed_33_axes_0 = const()[name = string("normed_33_axes_0"), val = tensor([-1])]; + fp16 var_1882_to_fp16 = const()[name = string("op_1882_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_33_cast_fp16 = layer_norm(axes = normed_33_axes_0, epsilon = var_1882_to_fp16, x = input_37_cast_fp16)[name = string("normed_33_cast_fp16")]; + tensor normed_35_begin_0 = const()[name = string("normed_35_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_35_end_0 = const()[name = string("normed_35_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_35_end_mask_0 = const()[name = string("normed_35_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_35_cast_fp16 = slice_by_index(begin = normed_35_begin_0, end = normed_35_end_0, end_mask = normed_35_end_mask_0, x = normed_33_cast_fp16)[name = string("normed_35_cast_fp16")]; + tensor const_72_promoted_to_fp16 = const()[name = string("const_72_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672166336)))]; + tensor hidden_states_23_cast_fp16 = mul(x = normed_35_cast_fp16, y = const_72_promoted_to_fp16)[name = string("hidden_states_23_cast_fp16")]; + tensor var_1910 = const()[name = string("op_1910"), val = tensor([0, 2, 1])]; + tensor var_1913_axes_0 = const()[name = string("op_1913_axes_0"), val = tensor([2])]; + tensor var_1911_cast_fp16 = transpose(perm = var_1910, x = hidden_states_23_cast_fp16)[name = string("transpose_107")]; + tensor var_1913_cast_fp16 = expand_dims(axes = var_1913_axes_0, x = var_1911_cast_fp16)[name = string("op_1913_cast_fp16")]; + string query_states_17_pad_type_0 = const()[name = string("query_states_17_pad_type_0"), val = string("valid")]; + tensor query_states_17_strides_0 = const()[name = string("query_states_17_strides_0"), val = tensor([1, 1])]; + tensor query_states_17_pad_0 = const()[name = string("query_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_17_dilations_0 = const()[name = string("query_states_17_dilations_0"), val = tensor([1, 1])]; + int32 query_states_17_groups_0 = const()[name = string("query_states_17_groups_0"), val = int32(1)]; + tensor query_states_17 = conv(dilations = query_states_17_dilations_0, groups = query_states_17_groups_0, pad = query_states_17_pad_0, pad_type = query_states_17_pad_type_0, strides = query_states_17_strides_0, weight = model_model_layers_16_self_attn_q_proj_weight_palettized, x = var_1913_cast_fp16)[name = string("query_states_17")]; + string key_states_21_pad_type_0 = const()[name = string("key_states_21_pad_type_0"), val = string("valid")]; + tensor key_states_21_strides_0 = const()[name = string("key_states_21_strides_0"), val = tensor([1, 1])]; + tensor key_states_21_pad_0 = const()[name = string("key_states_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_21_dilations_0 = const()[name = string("key_states_21_dilations_0"), val = tensor([1, 1])]; + int32 key_states_21_groups_0 = const()[name = string("key_states_21_groups_0"), val = int32(1)]; + tensor key_states_21 = conv(dilations = key_states_21_dilations_0, groups = key_states_21_groups_0, pad = key_states_21_pad_0, pad_type = key_states_21_pad_type_0, strides = key_states_21_strides_0, weight = model_model_layers_16_self_attn_k_proj_weight_palettized, x = var_1913_cast_fp16)[name = string("key_states_21")]; + string value_states_17_pad_type_0 = const()[name = string("value_states_17_pad_type_0"), val = string("valid")]; + tensor value_states_17_strides_0 = const()[name = string("value_states_17_strides_0"), val = tensor([1, 1])]; + tensor value_states_17_pad_0 = const()[name = string("value_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_17_dilations_0 = const()[name = string("value_states_17_dilations_0"), val = tensor([1, 1])]; + int32 value_states_17_groups_0 = const()[name = string("value_states_17_groups_0"), val = int32(1)]; + tensor value_states_17 = conv(dilations = value_states_17_dilations_0, groups = value_states_17_groups_0, pad = value_states_17_pad_0, pad_type = value_states_17_pad_type_0, strides = value_states_17_strides_0, weight = model_model_layers_16_self_attn_v_proj_weight_palettized, x = var_1913_cast_fp16)[name = string("value_states_17")]; + tensor var_1955 = const()[name = string("op_1955"), val = tensor([1, 16, 128, 128])]; + tensor var_1956 = reshape(shape = var_1955, x = query_states_17)[name = string("op_1956")]; + tensor var_1961 = const()[name = string("op_1961"), val = tensor([0, 1, 3, 2])]; + tensor var_1966 = const()[name = string("op_1966"), val = tensor([1, 8, 128, 128])]; + tensor var_1967 = reshape(shape = var_1966, x = key_states_21)[name = string("op_1967")]; + tensor var_1972 = const()[name = string("op_1972"), val = tensor([0, 1, 3, 2])]; + tensor var_1977 = const()[name = string("op_1977"), val = tensor([1, 8, 128, 128])]; + tensor var_1978 = reshape(shape = var_1977, x = value_states_17)[name = string("op_1978")]; + tensor var_1983 = const()[name = string("op_1983"), val = tensor([0, 1, 3, 2])]; + int32 var_1994 = const()[name = string("op_1994"), val = int32(-1)]; + fp16 const_74_promoted = const()[name = string("const_74_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_25 = transpose(perm = var_1961, x = var_1956)[name = string("transpose_106")]; + tensor var_1996 = mul(x = hidden_states_25, y = const_74_promoted)[name = string("op_1996")]; + bool input_41_interleave_0 = const()[name = string("input_41_interleave_0"), val = bool(false)]; + tensor input_41 = concat(axis = var_1994, interleave = input_41_interleave_0, values = (hidden_states_25, var_1996))[name = string("input_41")]; + tensor normed_37_axes_0 = const()[name = string("normed_37_axes_0"), val = tensor([-1])]; + fp16 var_1991_to_fp16 = const()[name = string("op_1991_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_37_cast_fp16 = layer_norm(axes = normed_37_axes_0, epsilon = var_1991_to_fp16, x = input_41)[name = string("normed_37_cast_fp16")]; + tensor normed_39_begin_0 = const()[name = string("normed_39_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_39_end_0 = const()[name = string("normed_39_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_39_end_mask_0 = const()[name = string("normed_39_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_39 = slice_by_index(begin = normed_39_begin_0, end = normed_39_end_0, end_mask = normed_39_end_mask_0, x = normed_37_cast_fp16)[name = string("normed_39")]; + tensor const_77 = const()[name = string("const_77"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672170496)))]; + tensor q_5 = mul(x = normed_39, y = const_77)[name = string("q_5")]; + int32 var_2019 = const()[name = string("op_2019"), val = int32(-1)]; + fp16 const_78_promoted = const()[name = string("const_78_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_27 = transpose(perm = var_1972, x = var_1967)[name = string("transpose_105")]; + tensor var_2021 = mul(x = hidden_states_27, y = const_78_promoted)[name = string("op_2021")]; + bool input_43_interleave_0 = const()[name = string("input_43_interleave_0"), val = bool(false)]; + tensor input_43 = concat(axis = var_2019, interleave = input_43_interleave_0, values = (hidden_states_27, var_2021))[name = string("input_43")]; + tensor normed_41_axes_0 = const()[name = string("normed_41_axes_0"), val = tensor([-1])]; + fp16 var_2016_to_fp16 = const()[name = string("op_2016_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_41_cast_fp16 = layer_norm(axes = normed_41_axes_0, epsilon = var_2016_to_fp16, x = input_43)[name = string("normed_41_cast_fp16")]; + tensor normed_43_begin_0 = const()[name = string("normed_43_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_43_end_0 = const()[name = string("normed_43_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_43_end_mask_0 = const()[name = string("normed_43_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_43 = slice_by_index(begin = normed_43_begin_0, end = normed_43_end_0, end_mask = normed_43_end_mask_0, x = normed_41_cast_fp16)[name = string("normed_43")]; + tensor const_81 = const()[name = string("const_81"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672170816)))]; + tensor k_5 = mul(x = normed_43, y = const_81)[name = string("k_5")]; + tensor var_2047 = mul(x = q_5, y = cos_5)[name = string("op_2047")]; + tensor x1_9_begin_0 = const()[name = string("x1_9_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_9_end_0 = const()[name = string("x1_9_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_9_end_mask_0 = const()[name = string("x1_9_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_9 = slice_by_index(begin = x1_9_begin_0, end = x1_9_end_0, end_mask = x1_9_end_mask_0, x = q_5)[name = string("x1_9")]; + tensor x2_9_begin_0 = const()[name = string("x2_9_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_9_end_0 = const()[name = string("x2_9_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_9_end_mask_0 = const()[name = string("x2_9_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_9 = slice_by_index(begin = x2_9_begin_0, end = x2_9_end_0, end_mask = x2_9_end_mask_0, x = q_5)[name = string("x2_9")]; + fp16 const_84_promoted = const()[name = string("const_84_promoted"), val = fp16(-0x1p+0)]; + tensor var_2068 = mul(x = x2_9, y = const_84_promoted)[name = string("op_2068")]; + int32 var_2070 = const()[name = string("op_2070"), val = int32(-1)]; + bool var_2071_interleave_0 = const()[name = string("op_2071_interleave_0"), val = bool(false)]; + tensor var_2071 = concat(axis = var_2070, interleave = var_2071_interleave_0, values = (var_2068, x1_9))[name = string("op_2071")]; + tensor var_2072 = mul(x = var_2071, y = sin_5)[name = string("op_2072")]; + tensor query_states_19 = add(x = var_2047, y = var_2072)[name = string("query_states_19")]; + tensor var_2075 = mul(x = k_5, y = cos_5)[name = string("op_2075")]; + tensor x1_11_begin_0 = const()[name = string("x1_11_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_11_end_0 = const()[name = string("x1_11_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_11_end_mask_0 = const()[name = string("x1_11_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_11 = slice_by_index(begin = x1_11_begin_0, end = x1_11_end_0, end_mask = x1_11_end_mask_0, x = k_5)[name = string("x1_11")]; + tensor x2_11_begin_0 = const()[name = string("x2_11_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_11_end_0 = const()[name = string("x2_11_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_11_end_mask_0 = const()[name = string("x2_11_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_11 = slice_by_index(begin = x2_11_begin_0, end = x2_11_end_0, end_mask = x2_11_end_mask_0, x = k_5)[name = string("x2_11")]; + fp16 const_87_promoted = const()[name = string("const_87_promoted"), val = fp16(-0x1p+0)]; + tensor var_2096 = mul(x = x2_11, y = const_87_promoted)[name = string("op_2096")]; + int32 var_2098 = const()[name = string("op_2098"), val = int32(-1)]; + bool var_2099_interleave_0 = const()[name = string("op_2099_interleave_0"), val = bool(false)]; + tensor var_2099 = concat(axis = var_2098, interleave = var_2099_interleave_0, values = (var_2096, x1_11))[name = string("op_2099")]; + tensor var_2100 = mul(x = var_2099, y = sin_5)[name = string("op_2100")]; + tensor key_states_23 = add(x = var_2075, y = var_2100)[name = string("key_states_23")]; + tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([16])]; + tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; + tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; + tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([17])]; + int32 concat_38_axis_0 = const()[name = string("concat_38_axis_0"), val = int32(0)]; + bool concat_38_interleave_0 = const()[name = string("concat_38_interleave_0"), val = bool(false)]; + tensor concat_38 = concat(axis = concat_38_axis_0, interleave = concat_38_interleave_0, values = (expand_dims_24, expand_dims_25, current_pos, expand_dims_27))[name = string("concat_38")]; + tensor concat_39_values1_0 = const()[name = string("concat_39_values1_0"), val = tensor([0])]; + tensor concat_39_values3_0 = const()[name = string("concat_39_values3_0"), val = tensor([0])]; + int32 concat_39_axis_0 = const()[name = string("concat_39_axis_0"), val = int32(0)]; + bool concat_39_interleave_0 = const()[name = string("concat_39_interleave_0"), val = bool(false)]; + tensor concat_39 = concat(axis = concat_39_axis_0, interleave = concat_39_interleave_0, values = (expand_dims_28, concat_39_values1_0, var_1042, concat_39_values3_0))[name = string("concat_39")]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_38, begin_mask = model_model_kv_cache_0_internal_tensor_assign_5_begin_mask_0, end = concat_39, end_mask = model_model_kv_cache_0_internal_tensor_assign_5_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_5_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_5_stride_0, update = key_states_23, x = coreml_update_state_31)[name = string("model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_5_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_88_write_state")]; + tensor coreml_update_state_32 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_88")]; + tensor expand_dims_30 = const()[name = string("expand_dims_30"), val = tensor([44])]; + tensor expand_dims_31 = const()[name = string("expand_dims_31"), val = tensor([0])]; + tensor expand_dims_33 = const()[name = string("expand_dims_33"), val = tensor([0])]; + tensor expand_dims_34 = const()[name = string("expand_dims_34"), val = tensor([45])]; + int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; + bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; + tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_30, expand_dims_31, current_pos, expand_dims_33))[name = string("concat_42")]; + tensor concat_43_values1_0 = const()[name = string("concat_43_values1_0"), val = tensor([0])]; + tensor concat_43_values3_0 = const()[name = string("concat_43_values3_0"), val = tensor([0])]; + int32 concat_43_axis_0 = const()[name = string("concat_43_axis_0"), val = int32(0)]; + bool concat_43_interleave_0 = const()[name = string("concat_43_interleave_0"), val = bool(false)]; + tensor concat_43 = concat(axis = concat_43_axis_0, interleave = concat_43_interleave_0, values = (expand_dims_34, concat_43_values1_0, var_1042, concat_43_values3_0))[name = string("concat_43")]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_19 = transpose(perm = var_1983, x = var_1978)[name = string("transpose_104")]; + tensor model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_42, begin_mask = model_model_kv_cache_0_internal_tensor_assign_6_begin_mask_0, end = concat_43, end_mask = model_model_kv_cache_0_internal_tensor_assign_6_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_6_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_6_stride_0, update = value_states_19, x = coreml_update_state_32)[name = string("model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_6_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_89_write_state")]; + tensor coreml_update_state_33 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_89")]; + tensor var_2171_begin_0 = const()[name = string("op_2171_begin_0"), val = tensor([16, 0, 0, 0])]; + tensor var_2171_end_0 = const()[name = string("op_2171_end_0"), val = tensor([17, 8, 1024, 128])]; + tensor var_2171_end_mask_0 = const()[name = string("op_2171_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2171_cast_fp16 = slice_by_index(begin = var_2171_begin_0, end = var_2171_end_0, end_mask = var_2171_end_mask_0, x = coreml_update_state_33)[name = string("op_2171_cast_fp16")]; + tensor K_layer_cache_5_axes_0 = const()[name = string("K_layer_cache_5_axes_0"), val = tensor([0])]; + tensor K_layer_cache_5_cast_fp16 = squeeze(axes = K_layer_cache_5_axes_0, x = var_2171_cast_fp16)[name = string("K_layer_cache_5_cast_fp16")]; + tensor var_2178_begin_0 = const()[name = string("op_2178_begin_0"), val = tensor([44, 0, 0, 0])]; + tensor var_2178_end_0 = const()[name = string("op_2178_end_0"), val = tensor([45, 8, 1024, 128])]; + tensor var_2178_end_mask_0 = const()[name = string("op_2178_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2178_cast_fp16 = slice_by_index(begin = var_2178_begin_0, end = var_2178_end_0, end_mask = var_2178_end_mask_0, x = coreml_update_state_33)[name = string("op_2178_cast_fp16")]; + tensor V_layer_cache_5_axes_0 = const()[name = string("V_layer_cache_5_axes_0"), val = tensor([0])]; + tensor V_layer_cache_5_cast_fp16 = squeeze(axes = V_layer_cache_5_axes_0, x = var_2178_cast_fp16)[name = string("V_layer_cache_5_cast_fp16")]; + tensor x_35_axes_0 = const()[name = string("x_35_axes_0"), val = tensor([1])]; + tensor x_35_cast_fp16 = expand_dims(axes = x_35_axes_0, x = K_layer_cache_5_cast_fp16)[name = string("x_35_cast_fp16")]; + tensor var_2207 = const()[name = string("op_2207"), val = tensor([1, 2, 1, 1])]; + tensor x_37_cast_fp16 = tile(reps = var_2207, x = x_35_cast_fp16)[name = string("x_37_cast_fp16")]; + tensor var_2219 = const()[name = string("op_2219"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_27_cast_fp16 = reshape(shape = var_2219, x = x_37_cast_fp16)[name = string("key_states_27_cast_fp16")]; + tensor x_41_axes_0 = const()[name = string("x_41_axes_0"), val = tensor([1])]; + tensor x_41_cast_fp16 = expand_dims(axes = x_41_axes_0, x = V_layer_cache_5_cast_fp16)[name = string("x_41_cast_fp16")]; + tensor var_2227 = const()[name = string("op_2227"), val = tensor([1, 2, 1, 1])]; + tensor x_43_cast_fp16 = tile(reps = var_2227, x = x_41_cast_fp16)[name = string("x_43_cast_fp16")]; + bool var_2254_transpose_x_0 = const()[name = string("op_2254_transpose_x_0"), val = bool(false)]; + bool var_2254_transpose_y_0 = const()[name = string("op_2254_transpose_y_0"), val = bool(true)]; + tensor var_2254 = matmul(transpose_x = var_2254_transpose_x_0, transpose_y = var_2254_transpose_y_0, x = query_states_19, y = key_states_27_cast_fp16)[name = string("op_2254")]; + fp16 var_2255_to_fp16 = const()[name = string("op_2255_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_9_cast_fp16 = mul(x = var_2254, y = var_2255_to_fp16)[name = string("attn_weights_9_cast_fp16")]; + tensor attn_weights_11_cast_fp16 = add(x = attn_weights_9_cast_fp16, y = causal_mask)[name = string("attn_weights_11_cast_fp16")]; + int32 var_2290 = const()[name = string("op_2290"), val = int32(-1)]; + tensor var_2292_cast_fp16 = softmax(axis = var_2290, x = attn_weights_11_cast_fp16)[name = string("op_2292_cast_fp16")]; + tensor concat_48 = const()[name = string("concat_48"), val = tensor([16, 128, 1024])]; + tensor reshape_6_cast_fp16 = reshape(shape = concat_48, x = var_2292_cast_fp16)[name = string("reshape_6_cast_fp16")]; + tensor concat_49 = const()[name = string("concat_49"), val = tensor([16, 1024, 128])]; + tensor reshape_7_cast_fp16 = reshape(shape = concat_49, x = x_43_cast_fp16)[name = string("reshape_7_cast_fp16")]; + bool matmul_2_transpose_x_0 = const()[name = string("matmul_2_transpose_x_0"), val = bool(false)]; + bool matmul_2_transpose_y_0 = const()[name = string("matmul_2_transpose_y_0"), val = bool(false)]; + tensor matmul_2_cast_fp16 = matmul(transpose_x = matmul_2_transpose_x_0, transpose_y = matmul_2_transpose_y_0, x = reshape_6_cast_fp16, y = reshape_7_cast_fp16)[name = string("matmul_2_cast_fp16")]; + tensor concat_53 = const()[name = string("concat_53"), val = tensor([1, 16, 128, 128])]; + tensor reshape_8_cast_fp16 = reshape(shape = concat_53, x = matmul_2_cast_fp16)[name = string("reshape_8_cast_fp16")]; + tensor var_2304_perm_0 = const()[name = string("op_2304_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_2323 = const()[name = string("op_2323"), val = tensor([1, 128, 2048])]; + tensor var_2304_cast_fp16 = transpose(perm = var_2304_perm_0, x = reshape_8_cast_fp16)[name = string("transpose_103")]; + tensor attn_output_25_cast_fp16 = reshape(shape = var_2323, x = var_2304_cast_fp16)[name = string("attn_output_25_cast_fp16")]; + tensor var_2328 = const()[name = string("op_2328"), val = tensor([0, 2, 1])]; + string var_2344_pad_type_0 = const()[name = string("op_2344_pad_type_0"), val = string("valid")]; + int32 var_2344_groups_0 = const()[name = string("op_2344_groups_0"), val = int32(1)]; + tensor var_2344_strides_0 = const()[name = string("op_2344_strides_0"), val = tensor([1])]; + tensor var_2344_pad_0 = const()[name = string("op_2344_pad_0"), val = tensor([0, 0])]; + tensor var_2344_dilations_0 = const()[name = string("op_2344_dilations_0"), val = tensor([1])]; + tensor squeeze_2_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672171136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676365504))))[name = string("squeeze_2_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_2329_cast_fp16 = transpose(perm = var_2328, x = attn_output_25_cast_fp16)[name = string("transpose_102")]; + tensor var_2344_cast_fp16 = conv(dilations = var_2344_dilations_0, groups = var_2344_groups_0, pad = var_2344_pad_0, pad_type = var_2344_pad_type_0, strides = var_2344_strides_0, weight = squeeze_2_cast_fp16_to_fp32_to_fp16_palettized, x = var_2329_cast_fp16)[name = string("op_2344_cast_fp16")]; + tensor var_2348 = const()[name = string("op_2348"), val = tensor([0, 2, 1])]; + tensor attn_output_29_cast_fp16 = transpose(perm = var_2348, x = var_2344_cast_fp16)[name = string("transpose_101")]; + tensor hidden_states_29_cast_fp16 = add(x = hidden_states_21_cast_fp16, y = attn_output_29_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; + int32 var_2361 = const()[name = string("op_2361"), val = int32(-1)]; + fp16 const_99_promoted_to_fp16 = const()[name = string("const_99_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2363_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_99_promoted_to_fp16)[name = string("op_2363_cast_fp16")]; + bool input_47_interleave_0 = const()[name = string("input_47_interleave_0"), val = bool(false)]; + tensor input_47_cast_fp16 = concat(axis = var_2361, interleave = input_47_interleave_0, values = (hidden_states_29_cast_fp16, var_2363_cast_fp16))[name = string("input_47_cast_fp16")]; + tensor normed_45_axes_0 = const()[name = string("normed_45_axes_0"), val = tensor([-1])]; + fp16 var_2358_to_fp16 = const()[name = string("op_2358_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_45_cast_fp16 = layer_norm(axes = normed_45_axes_0, epsilon = var_2358_to_fp16, x = input_47_cast_fp16)[name = string("normed_45_cast_fp16")]; + tensor normed_47_begin_0 = const()[name = string("normed_47_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_47_end_0 = const()[name = string("normed_47_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_47_end_mask_0 = const()[name = string("normed_47_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_47_cast_fp16 = slice_by_index(begin = normed_47_begin_0, end = normed_47_end_0, end_mask = normed_47_end_mask_0, x = normed_45_cast_fp16)[name = string("normed_47_cast_fp16")]; + tensor const_102_promoted_to_fp16 = const()[name = string("const_102_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676496640)))]; + tensor x_45_cast_fp16 = mul(x = normed_47_cast_fp16, y = const_102_promoted_to_fp16)[name = string("x_45_cast_fp16")]; + tensor var_2388 = const()[name = string("op_2388"), val = tensor([0, 2, 1])]; + tensor input_49_axes_0 = const()[name = string("input_49_axes_0"), val = tensor([2])]; + tensor var_2389 = transpose(perm = var_2388, x = x_45_cast_fp16)[name = string("transpose_100")]; + tensor input_49 = expand_dims(axes = input_49_axes_0, x = var_2389)[name = string("input_49")]; + string input_51_pad_type_0 = const()[name = string("input_51_pad_type_0"), val = string("valid")]; + tensor input_51_strides_0 = const()[name = string("input_51_strides_0"), val = tensor([1, 1])]; + tensor input_51_pad_0 = const()[name = string("input_51_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_51_dilations_0 = const()[name = string("input_51_dilations_0"), val = tensor([1, 1])]; + int32 input_51_groups_0 = const()[name = string("input_51_groups_0"), val = int32(1)]; + tensor input_51 = conv(dilations = input_51_dilations_0, groups = input_51_groups_0, pad = input_51_pad_0, pad_type = input_51_pad_type_0, strides = input_51_strides_0, weight = model_model_layers_16_mlp_gate_proj_weight_palettized, x = input_49)[name = string("input_51")]; + string b_5_pad_type_0 = const()[name = string("b_5_pad_type_0"), val = string("valid")]; + tensor b_5_strides_0 = const()[name = string("b_5_strides_0"), val = tensor([1, 1])]; + tensor b_5_pad_0 = const()[name = string("b_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_5_dilations_0 = const()[name = string("b_5_dilations_0"), val = tensor([1, 1])]; + int32 b_5_groups_0 = const()[name = string("b_5_groups_0"), val = int32(1)]; + tensor b_5 = conv(dilations = b_5_dilations_0, groups = b_5_groups_0, pad = b_5_pad_0, pad_type = b_5_pad_type_0, strides = b_5_strides_0, weight = model_model_layers_16_mlp_up_proj_weight_palettized, x = input_49)[name = string("b_5")]; + tensor c_5 = silu(x = input_51)[name = string("c_5")]; + tensor input_53 = mul(x = c_5, y = b_5)[name = string("input_53")]; + string e_5_pad_type_0 = const()[name = string("e_5_pad_type_0"), val = string("valid")]; + tensor e_5_strides_0 = const()[name = string("e_5_strides_0"), val = tensor([1, 1])]; + tensor e_5_pad_0 = const()[name = string("e_5_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_5_dilations_0 = const()[name = string("e_5_dilations_0"), val = tensor([1, 1])]; + int32 e_5_groups_0 = const()[name = string("e_5_groups_0"), val = int32(1)]; + tensor e_5 = conv(dilations = e_5_dilations_0, groups = e_5_groups_0, pad = e_5_pad_0, pad_type = e_5_pad_type_0, strides = e_5_strides_0, weight = model_model_layers_16_mlp_down_proj_weight_palettized, x = input_53)[name = string("e_5")]; + tensor var_2411_axes_0 = const()[name = string("op_2411_axes_0"), val = tensor([2])]; + tensor var_2411 = squeeze(axes = var_2411_axes_0, x = e_5)[name = string("op_2411")]; + tensor var_2412 = const()[name = string("op_2412"), val = tensor([0, 2, 1])]; + tensor var_2413 = transpose(perm = var_2412, x = var_2411)[name = string("transpose_99")]; + tensor hidden_states_31_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = var_2413)[name = string("hidden_states_31_cast_fp16")]; + int32 var_2425 = const()[name = string("op_2425"), val = int32(-1)]; + fp16 const_103_promoted_to_fp16 = const()[name = string("const_103_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2427_cast_fp16 = mul(x = hidden_states_31_cast_fp16, y = const_103_promoted_to_fp16)[name = string("op_2427_cast_fp16")]; + bool input_55_interleave_0 = const()[name = string("input_55_interleave_0"), val = bool(false)]; + tensor input_55_cast_fp16 = concat(axis = var_2425, interleave = input_55_interleave_0, values = (hidden_states_31_cast_fp16, var_2427_cast_fp16))[name = string("input_55_cast_fp16")]; + tensor normed_49_axes_0 = const()[name = string("normed_49_axes_0"), val = tensor([-1])]; + fp16 var_2422_to_fp16 = const()[name = string("op_2422_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_49_cast_fp16 = layer_norm(axes = normed_49_axes_0, epsilon = var_2422_to_fp16, x = input_55_cast_fp16)[name = string("normed_49_cast_fp16")]; + tensor normed_51_begin_0 = const()[name = string("normed_51_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_51_end_0 = const()[name = string("normed_51_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_51_end_mask_0 = const()[name = string("normed_51_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_51_cast_fp16 = slice_by_index(begin = normed_51_begin_0, end = normed_51_end_0, end_mask = normed_51_end_mask_0, x = normed_49_cast_fp16)[name = string("normed_51_cast_fp16")]; + tensor const_106_promoted_to_fp16 = const()[name = string("const_106_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676500800)))]; + tensor hidden_states_33_cast_fp16 = mul(x = normed_51_cast_fp16, y = const_106_promoted_to_fp16)[name = string("hidden_states_33_cast_fp16")]; + tensor var_2450 = const()[name = string("op_2450"), val = tensor([0, 2, 1])]; + tensor var_2453_axes_0 = const()[name = string("op_2453_axes_0"), val = tensor([2])]; + tensor var_2451_cast_fp16 = transpose(perm = var_2450, x = hidden_states_33_cast_fp16)[name = string("transpose_98")]; + tensor var_2453_cast_fp16 = expand_dims(axes = var_2453_axes_0, x = var_2451_cast_fp16)[name = string("op_2453_cast_fp16")]; + string query_states_25_pad_type_0 = const()[name = string("query_states_25_pad_type_0"), val = string("valid")]; + tensor query_states_25_strides_0 = const()[name = string("query_states_25_strides_0"), val = tensor([1, 1])]; + tensor query_states_25_pad_0 = const()[name = string("query_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_25_dilations_0 = const()[name = string("query_states_25_dilations_0"), val = tensor([1, 1])]; + int32 query_states_25_groups_0 = const()[name = string("query_states_25_groups_0"), val = int32(1)]; + tensor query_states_25 = conv(dilations = query_states_25_dilations_0, groups = query_states_25_groups_0, pad = query_states_25_pad_0, pad_type = query_states_25_pad_type_0, strides = query_states_25_strides_0, weight = model_model_layers_17_self_attn_q_proj_weight_palettized, x = var_2453_cast_fp16)[name = string("query_states_25")]; + string key_states_31_pad_type_0 = const()[name = string("key_states_31_pad_type_0"), val = string("valid")]; + tensor key_states_31_strides_0 = const()[name = string("key_states_31_strides_0"), val = tensor([1, 1])]; + tensor key_states_31_pad_0 = const()[name = string("key_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_31_dilations_0 = const()[name = string("key_states_31_dilations_0"), val = tensor([1, 1])]; + int32 key_states_31_groups_0 = const()[name = string("key_states_31_groups_0"), val = int32(1)]; + tensor key_states_31 = conv(dilations = key_states_31_dilations_0, groups = key_states_31_groups_0, pad = key_states_31_pad_0, pad_type = key_states_31_pad_type_0, strides = key_states_31_strides_0, weight = model_model_layers_17_self_attn_k_proj_weight_palettized, x = var_2453_cast_fp16)[name = string("key_states_31")]; + string value_states_25_pad_type_0 = const()[name = string("value_states_25_pad_type_0"), val = string("valid")]; + tensor value_states_25_strides_0 = const()[name = string("value_states_25_strides_0"), val = tensor([1, 1])]; + tensor value_states_25_pad_0 = const()[name = string("value_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_25_dilations_0 = const()[name = string("value_states_25_dilations_0"), val = tensor([1, 1])]; + int32 value_states_25_groups_0 = const()[name = string("value_states_25_groups_0"), val = int32(1)]; + tensor value_states_25 = conv(dilations = value_states_25_dilations_0, groups = value_states_25_groups_0, pad = value_states_25_pad_0, pad_type = value_states_25_pad_type_0, strides = value_states_25_strides_0, weight = model_model_layers_17_self_attn_v_proj_weight_palettized, x = var_2453_cast_fp16)[name = string("value_states_25")]; + tensor var_2495 = const()[name = string("op_2495"), val = tensor([1, 16, 128, 128])]; + tensor var_2496 = reshape(shape = var_2495, x = query_states_25)[name = string("op_2496")]; + tensor var_2501 = const()[name = string("op_2501"), val = tensor([0, 1, 3, 2])]; + tensor var_2506 = const()[name = string("op_2506"), val = tensor([1, 8, 128, 128])]; + tensor var_2507 = reshape(shape = var_2506, x = key_states_31)[name = string("op_2507")]; + tensor var_2512 = const()[name = string("op_2512"), val = tensor([0, 1, 3, 2])]; + tensor var_2517 = const()[name = string("op_2517"), val = tensor([1, 8, 128, 128])]; + tensor var_2518 = reshape(shape = var_2517, x = value_states_25)[name = string("op_2518")]; + tensor var_2523 = const()[name = string("op_2523"), val = tensor([0, 1, 3, 2])]; + int32 var_2534 = const()[name = string("op_2534"), val = int32(-1)]; + fp16 const_108_promoted = const()[name = string("const_108_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_35 = transpose(perm = var_2501, x = var_2496)[name = string("transpose_97")]; + tensor var_2536 = mul(x = hidden_states_35, y = const_108_promoted)[name = string("op_2536")]; + bool input_59_interleave_0 = const()[name = string("input_59_interleave_0"), val = bool(false)]; + tensor input_59 = concat(axis = var_2534, interleave = input_59_interleave_0, values = (hidden_states_35, var_2536))[name = string("input_59")]; + tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; + fp16 var_2531_to_fp16 = const()[name = string("op_2531_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_2531_to_fp16, x = input_59)[name = string("normed_53_cast_fp16")]; + tensor normed_55_begin_0 = const()[name = string("normed_55_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_55_end_0 = const()[name = string("normed_55_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_55_end_mask_0 = const()[name = string("normed_55_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_55 = slice_by_index(begin = normed_55_begin_0, end = normed_55_end_0, end_mask = normed_55_end_mask_0, x = normed_53_cast_fp16)[name = string("normed_55")]; + tensor const_111 = const()[name = string("const_111"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676504960)))]; + tensor q_7 = mul(x = normed_55, y = const_111)[name = string("q_7")]; + int32 var_2559 = const()[name = string("op_2559"), val = int32(-1)]; + fp16 const_112_promoted = const()[name = string("const_112_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_37 = transpose(perm = var_2512, x = var_2507)[name = string("transpose_96")]; + tensor var_2561 = mul(x = hidden_states_37, y = const_112_promoted)[name = string("op_2561")]; + bool input_61_interleave_0 = const()[name = string("input_61_interleave_0"), val = bool(false)]; + tensor input_61 = concat(axis = var_2559, interleave = input_61_interleave_0, values = (hidden_states_37, var_2561))[name = string("input_61")]; + tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; + fp16 var_2556_to_fp16 = const()[name = string("op_2556_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_2556_to_fp16, x = input_61)[name = string("normed_57_cast_fp16")]; + tensor normed_59_begin_0 = const()[name = string("normed_59_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_59_end_0 = const()[name = string("normed_59_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_59_end_mask_0 = const()[name = string("normed_59_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_59 = slice_by_index(begin = normed_59_begin_0, end = normed_59_end_0, end_mask = normed_59_end_mask_0, x = normed_57_cast_fp16)[name = string("normed_59")]; + tensor const_115 = const()[name = string("const_115"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676505280)))]; + tensor k_7 = mul(x = normed_59, y = const_115)[name = string("k_7")]; + tensor var_2587 = mul(x = q_7, y = cos_5)[name = string("op_2587")]; + tensor x1_13_begin_0 = const()[name = string("x1_13_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_13_end_0 = const()[name = string("x1_13_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_13_end_mask_0 = const()[name = string("x1_13_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_13 = slice_by_index(begin = x1_13_begin_0, end = x1_13_end_0, end_mask = x1_13_end_mask_0, x = q_7)[name = string("x1_13")]; + tensor x2_13_begin_0 = const()[name = string("x2_13_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_13_end_0 = const()[name = string("x2_13_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_13_end_mask_0 = const()[name = string("x2_13_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_13 = slice_by_index(begin = x2_13_begin_0, end = x2_13_end_0, end_mask = x2_13_end_mask_0, x = q_7)[name = string("x2_13")]; + fp16 const_118_promoted = const()[name = string("const_118_promoted"), val = fp16(-0x1p+0)]; + tensor var_2608 = mul(x = x2_13, y = const_118_promoted)[name = string("op_2608")]; + int32 var_2610 = const()[name = string("op_2610"), val = int32(-1)]; + bool var_2611_interleave_0 = const()[name = string("op_2611_interleave_0"), val = bool(false)]; + tensor var_2611 = concat(axis = var_2610, interleave = var_2611_interleave_0, values = (var_2608, x1_13))[name = string("op_2611")]; + tensor var_2612 = mul(x = var_2611, y = sin_5)[name = string("op_2612")]; + tensor query_states_27 = add(x = var_2587, y = var_2612)[name = string("query_states_27")]; + tensor var_2615 = mul(x = k_7, y = cos_5)[name = string("op_2615")]; + tensor x1_15_begin_0 = const()[name = string("x1_15_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_15_end_0 = const()[name = string("x1_15_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_15_end_mask_0 = const()[name = string("x1_15_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_15 = slice_by_index(begin = x1_15_begin_0, end = x1_15_end_0, end_mask = x1_15_end_mask_0, x = k_7)[name = string("x1_15")]; + tensor x2_15_begin_0 = const()[name = string("x2_15_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_15_end_0 = const()[name = string("x2_15_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_15_end_mask_0 = const()[name = string("x2_15_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_15 = slice_by_index(begin = x2_15_begin_0, end = x2_15_end_0, end_mask = x2_15_end_mask_0, x = k_7)[name = string("x2_15")]; + fp16 const_121_promoted = const()[name = string("const_121_promoted"), val = fp16(-0x1p+0)]; + tensor var_2636 = mul(x = x2_15, y = const_121_promoted)[name = string("op_2636")]; + int32 var_2638 = const()[name = string("op_2638"), val = int32(-1)]; + bool var_2639_interleave_0 = const()[name = string("op_2639_interleave_0"), val = bool(false)]; + tensor var_2639 = concat(axis = var_2638, interleave = var_2639_interleave_0, values = (var_2636, x1_15))[name = string("op_2639")]; + tensor var_2640 = mul(x = var_2639, y = sin_5)[name = string("op_2640")]; + tensor key_states_33 = add(x = var_2615, y = var_2640)[name = string("key_states_33")]; + tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([17])]; + tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; + tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; + tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([18])]; + int32 concat_56_axis_0 = const()[name = string("concat_56_axis_0"), val = int32(0)]; + bool concat_56_interleave_0 = const()[name = string("concat_56_interleave_0"), val = bool(false)]; + tensor concat_56 = concat(axis = concat_56_axis_0, interleave = concat_56_interleave_0, values = (expand_dims_36, expand_dims_37, current_pos, expand_dims_39))[name = string("concat_56")]; + tensor concat_57_values1_0 = const()[name = string("concat_57_values1_0"), val = tensor([0])]; + tensor concat_57_values3_0 = const()[name = string("concat_57_values3_0"), val = tensor([0])]; + int32 concat_57_axis_0 = const()[name = string("concat_57_axis_0"), val = int32(0)]; + bool concat_57_interleave_0 = const()[name = string("concat_57_interleave_0"), val = bool(false)]; + tensor concat_57 = concat(axis = concat_57_axis_0, interleave = concat_57_interleave_0, values = (expand_dims_40, concat_57_values1_0, var_1042, concat_57_values3_0))[name = string("concat_57")]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_56, begin_mask = model_model_kv_cache_0_internal_tensor_assign_7_begin_mask_0, end = concat_57, end_mask = model_model_kv_cache_0_internal_tensor_assign_7_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_7_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_7_stride_0, update = key_states_33, x = coreml_update_state_33)[name = string("model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_7_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_90_write_state")]; + tensor coreml_update_state_34 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_90")]; + tensor expand_dims_42 = const()[name = string("expand_dims_42"), val = tensor([45])]; + tensor expand_dims_43 = const()[name = string("expand_dims_43"), val = tensor([0])]; + tensor expand_dims_45 = const()[name = string("expand_dims_45"), val = tensor([0])]; + tensor expand_dims_46 = const()[name = string("expand_dims_46"), val = tensor([46])]; + int32 concat_60_axis_0 = const()[name = string("concat_60_axis_0"), val = int32(0)]; + bool concat_60_interleave_0 = const()[name = string("concat_60_interleave_0"), val = bool(false)]; + tensor concat_60 = concat(axis = concat_60_axis_0, interleave = concat_60_interleave_0, values = (expand_dims_42, expand_dims_43, current_pos, expand_dims_45))[name = string("concat_60")]; + tensor concat_61_values1_0 = const()[name = string("concat_61_values1_0"), val = tensor([0])]; + tensor concat_61_values3_0 = const()[name = string("concat_61_values3_0"), val = tensor([0])]; + int32 concat_61_axis_0 = const()[name = string("concat_61_axis_0"), val = int32(0)]; + bool concat_61_interleave_0 = const()[name = string("concat_61_interleave_0"), val = bool(false)]; + tensor concat_61 = concat(axis = concat_61_axis_0, interleave = concat_61_interleave_0, values = (expand_dims_46, concat_61_values1_0, var_1042, concat_61_values3_0))[name = string("concat_61")]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_27 = transpose(perm = var_2523, x = var_2518)[name = string("transpose_95")]; + tensor model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_60, begin_mask = model_model_kv_cache_0_internal_tensor_assign_8_begin_mask_0, end = concat_61, end_mask = model_model_kv_cache_0_internal_tensor_assign_8_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_8_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_8_stride_0, update = value_states_27, x = coreml_update_state_34)[name = string("model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_8_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_91_write_state")]; + tensor coreml_update_state_35 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_91")]; + tensor var_2711_begin_0 = const()[name = string("op_2711_begin_0"), val = tensor([17, 0, 0, 0])]; + tensor var_2711_end_0 = const()[name = string("op_2711_end_0"), val = tensor([18, 8, 1024, 128])]; + tensor var_2711_end_mask_0 = const()[name = string("op_2711_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2711_cast_fp16 = slice_by_index(begin = var_2711_begin_0, end = var_2711_end_0, end_mask = var_2711_end_mask_0, x = coreml_update_state_35)[name = string("op_2711_cast_fp16")]; + tensor K_layer_cache_7_axes_0 = const()[name = string("K_layer_cache_7_axes_0"), val = tensor([0])]; + tensor K_layer_cache_7_cast_fp16 = squeeze(axes = K_layer_cache_7_axes_0, x = var_2711_cast_fp16)[name = string("K_layer_cache_7_cast_fp16")]; + tensor var_2718_begin_0 = const()[name = string("op_2718_begin_0"), val = tensor([45, 0, 0, 0])]; + tensor var_2718_end_0 = const()[name = string("op_2718_end_0"), val = tensor([46, 8, 1024, 128])]; + tensor var_2718_end_mask_0 = const()[name = string("op_2718_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_2718_cast_fp16 = slice_by_index(begin = var_2718_begin_0, end = var_2718_end_0, end_mask = var_2718_end_mask_0, x = coreml_update_state_35)[name = string("op_2718_cast_fp16")]; + tensor V_layer_cache_7_axes_0 = const()[name = string("V_layer_cache_7_axes_0"), val = tensor([0])]; + tensor V_layer_cache_7_cast_fp16 = squeeze(axes = V_layer_cache_7_axes_0, x = var_2718_cast_fp16)[name = string("V_layer_cache_7_cast_fp16")]; + tensor x_51_axes_0 = const()[name = string("x_51_axes_0"), val = tensor([1])]; + tensor x_51_cast_fp16 = expand_dims(axes = x_51_axes_0, x = K_layer_cache_7_cast_fp16)[name = string("x_51_cast_fp16")]; + tensor var_2747 = const()[name = string("op_2747"), val = tensor([1, 2, 1, 1])]; + tensor x_53_cast_fp16 = tile(reps = var_2747, x = x_51_cast_fp16)[name = string("x_53_cast_fp16")]; + tensor var_2759 = const()[name = string("op_2759"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_37_cast_fp16 = reshape(shape = var_2759, x = x_53_cast_fp16)[name = string("key_states_37_cast_fp16")]; + tensor x_57_axes_0 = const()[name = string("x_57_axes_0"), val = tensor([1])]; + tensor x_57_cast_fp16 = expand_dims(axes = x_57_axes_0, x = V_layer_cache_7_cast_fp16)[name = string("x_57_cast_fp16")]; + tensor var_2767 = const()[name = string("op_2767"), val = tensor([1, 2, 1, 1])]; + tensor x_59_cast_fp16 = tile(reps = var_2767, x = x_57_cast_fp16)[name = string("x_59_cast_fp16")]; + bool var_2794_transpose_x_0 = const()[name = string("op_2794_transpose_x_0"), val = bool(false)]; + bool var_2794_transpose_y_0 = const()[name = string("op_2794_transpose_y_0"), val = bool(true)]; + tensor var_2794 = matmul(transpose_x = var_2794_transpose_x_0, transpose_y = var_2794_transpose_y_0, x = query_states_27, y = key_states_37_cast_fp16)[name = string("op_2794")]; + fp16 var_2795_to_fp16 = const()[name = string("op_2795_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_13_cast_fp16 = mul(x = var_2794, y = var_2795_to_fp16)[name = string("attn_weights_13_cast_fp16")]; + tensor attn_weights_15_cast_fp16 = add(x = attn_weights_13_cast_fp16, y = causal_mask)[name = string("attn_weights_15_cast_fp16")]; + int32 var_2830 = const()[name = string("op_2830"), val = int32(-1)]; + tensor var_2832_cast_fp16 = softmax(axis = var_2830, x = attn_weights_15_cast_fp16)[name = string("op_2832_cast_fp16")]; + tensor concat_66 = const()[name = string("concat_66"), val = tensor([16, 128, 1024])]; + tensor reshape_9_cast_fp16 = reshape(shape = concat_66, x = var_2832_cast_fp16)[name = string("reshape_9_cast_fp16")]; + tensor concat_67 = const()[name = string("concat_67"), val = tensor([16, 1024, 128])]; + tensor reshape_10_cast_fp16 = reshape(shape = concat_67, x = x_59_cast_fp16)[name = string("reshape_10_cast_fp16")]; + bool matmul_3_transpose_x_0 = const()[name = string("matmul_3_transpose_x_0"), val = bool(false)]; + bool matmul_3_transpose_y_0 = const()[name = string("matmul_3_transpose_y_0"), val = bool(false)]; + tensor matmul_3_cast_fp16 = matmul(transpose_x = matmul_3_transpose_x_0, transpose_y = matmul_3_transpose_y_0, x = reshape_9_cast_fp16, y = reshape_10_cast_fp16)[name = string("matmul_3_cast_fp16")]; + tensor concat_71 = const()[name = string("concat_71"), val = tensor([1, 16, 128, 128])]; + tensor reshape_11_cast_fp16 = reshape(shape = concat_71, x = matmul_3_cast_fp16)[name = string("reshape_11_cast_fp16")]; + tensor var_2844_perm_0 = const()[name = string("op_2844_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_2863 = const()[name = string("op_2863"), val = tensor([1, 128, 2048])]; + tensor var_2844_cast_fp16 = transpose(perm = var_2844_perm_0, x = reshape_11_cast_fp16)[name = string("transpose_94")]; + tensor attn_output_35_cast_fp16 = reshape(shape = var_2863, x = var_2844_cast_fp16)[name = string("attn_output_35_cast_fp16")]; + tensor var_2868 = const()[name = string("op_2868"), val = tensor([0, 2, 1])]; + string var_2884_pad_type_0 = const()[name = string("op_2884_pad_type_0"), val = string("valid")]; + int32 var_2884_groups_0 = const()[name = string("op_2884_groups_0"), val = int32(1)]; + tensor var_2884_strides_0 = const()[name = string("op_2884_strides_0"), val = tensor([1])]; + tensor var_2884_pad_0 = const()[name = string("op_2884_pad_0"), val = tensor([0, 0])]; + tensor var_2884_dilations_0 = const()[name = string("op_2884_dilations_0"), val = tensor([1])]; + tensor squeeze_3_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(676505600))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680699968))))[name = string("squeeze_3_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_2869_cast_fp16 = transpose(perm = var_2868, x = attn_output_35_cast_fp16)[name = string("transpose_93")]; + tensor var_2884_cast_fp16 = conv(dilations = var_2884_dilations_0, groups = var_2884_groups_0, pad = var_2884_pad_0, pad_type = var_2884_pad_type_0, strides = var_2884_strides_0, weight = squeeze_3_cast_fp16_to_fp32_to_fp16_palettized, x = var_2869_cast_fp16)[name = string("op_2884_cast_fp16")]; + tensor var_2888 = const()[name = string("op_2888"), val = tensor([0, 2, 1])]; + tensor attn_output_39_cast_fp16 = transpose(perm = var_2888, x = var_2884_cast_fp16)[name = string("transpose_92")]; + tensor hidden_states_39_cast_fp16 = add(x = hidden_states_31_cast_fp16, y = attn_output_39_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; + int32 var_2901 = const()[name = string("op_2901"), val = int32(-1)]; + fp16 const_133_promoted_to_fp16 = const()[name = string("const_133_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2903_cast_fp16 = mul(x = hidden_states_39_cast_fp16, y = const_133_promoted_to_fp16)[name = string("op_2903_cast_fp16")]; + bool input_65_interleave_0 = const()[name = string("input_65_interleave_0"), val = bool(false)]; + tensor input_65_cast_fp16 = concat(axis = var_2901, interleave = input_65_interleave_0, values = (hidden_states_39_cast_fp16, var_2903_cast_fp16))[name = string("input_65_cast_fp16")]; + tensor normed_61_axes_0 = const()[name = string("normed_61_axes_0"), val = tensor([-1])]; + fp16 var_2898_to_fp16 = const()[name = string("op_2898_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_61_cast_fp16 = layer_norm(axes = normed_61_axes_0, epsilon = var_2898_to_fp16, x = input_65_cast_fp16)[name = string("normed_61_cast_fp16")]; + tensor normed_63_begin_0 = const()[name = string("normed_63_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_63_end_0 = const()[name = string("normed_63_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_63_end_mask_0 = const()[name = string("normed_63_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_63_cast_fp16 = slice_by_index(begin = normed_63_begin_0, end = normed_63_end_0, end_mask = normed_63_end_mask_0, x = normed_61_cast_fp16)[name = string("normed_63_cast_fp16")]; + tensor const_136_promoted_to_fp16 = const()[name = string("const_136_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680831104)))]; + tensor x_61_cast_fp16 = mul(x = normed_63_cast_fp16, y = const_136_promoted_to_fp16)[name = string("x_61_cast_fp16")]; + tensor var_2928 = const()[name = string("op_2928"), val = tensor([0, 2, 1])]; + tensor input_67_axes_0 = const()[name = string("input_67_axes_0"), val = tensor([2])]; + tensor var_2929 = transpose(perm = var_2928, x = x_61_cast_fp16)[name = string("transpose_91")]; + tensor input_67 = expand_dims(axes = input_67_axes_0, x = var_2929)[name = string("input_67")]; + string input_69_pad_type_0 = const()[name = string("input_69_pad_type_0"), val = string("valid")]; + tensor input_69_strides_0 = const()[name = string("input_69_strides_0"), val = tensor([1, 1])]; + tensor input_69_pad_0 = const()[name = string("input_69_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_69_dilations_0 = const()[name = string("input_69_dilations_0"), val = tensor([1, 1])]; + int32 input_69_groups_0 = const()[name = string("input_69_groups_0"), val = int32(1)]; + tensor input_69 = conv(dilations = input_69_dilations_0, groups = input_69_groups_0, pad = input_69_pad_0, pad_type = input_69_pad_type_0, strides = input_69_strides_0, weight = model_model_layers_17_mlp_gate_proj_weight_palettized, x = input_67)[name = string("input_69")]; + string b_7_pad_type_0 = const()[name = string("b_7_pad_type_0"), val = string("valid")]; + tensor b_7_strides_0 = const()[name = string("b_7_strides_0"), val = tensor([1, 1])]; + tensor b_7_pad_0 = const()[name = string("b_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_7_dilations_0 = const()[name = string("b_7_dilations_0"), val = tensor([1, 1])]; + int32 b_7_groups_0 = const()[name = string("b_7_groups_0"), val = int32(1)]; + tensor b_7 = conv(dilations = b_7_dilations_0, groups = b_7_groups_0, pad = b_7_pad_0, pad_type = b_7_pad_type_0, strides = b_7_strides_0, weight = model_model_layers_17_mlp_up_proj_weight_palettized, x = input_67)[name = string("b_7")]; + tensor c_7 = silu(x = input_69)[name = string("c_7")]; + tensor input_71 = mul(x = c_7, y = b_7)[name = string("input_71")]; + string e_7_pad_type_0 = const()[name = string("e_7_pad_type_0"), val = string("valid")]; + tensor e_7_strides_0 = const()[name = string("e_7_strides_0"), val = tensor([1, 1])]; + tensor e_7_pad_0 = const()[name = string("e_7_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_7_dilations_0 = const()[name = string("e_7_dilations_0"), val = tensor([1, 1])]; + int32 e_7_groups_0 = const()[name = string("e_7_groups_0"), val = int32(1)]; + tensor e_7 = conv(dilations = e_7_dilations_0, groups = e_7_groups_0, pad = e_7_pad_0, pad_type = e_7_pad_type_0, strides = e_7_strides_0, weight = model_model_layers_17_mlp_down_proj_weight_palettized, x = input_71)[name = string("e_7")]; + tensor var_2951_axes_0 = const()[name = string("op_2951_axes_0"), val = tensor([2])]; + tensor var_2951 = squeeze(axes = var_2951_axes_0, x = e_7)[name = string("op_2951")]; + tensor var_2952 = const()[name = string("op_2952"), val = tensor([0, 2, 1])]; + tensor var_2953 = transpose(perm = var_2952, x = var_2951)[name = string("transpose_90")]; + tensor hidden_states_41_cast_fp16 = add(x = hidden_states_39_cast_fp16, y = var_2953)[name = string("hidden_states_41_cast_fp16")]; + int32 var_2965 = const()[name = string("op_2965"), val = int32(-1)]; + fp16 const_137_promoted_to_fp16 = const()[name = string("const_137_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_2967_cast_fp16 = mul(x = hidden_states_41_cast_fp16, y = const_137_promoted_to_fp16)[name = string("op_2967_cast_fp16")]; + bool input_73_interleave_0 = const()[name = string("input_73_interleave_0"), val = bool(false)]; + tensor input_73_cast_fp16 = concat(axis = var_2965, interleave = input_73_interleave_0, values = (hidden_states_41_cast_fp16, var_2967_cast_fp16))[name = string("input_73_cast_fp16")]; + tensor normed_65_axes_0 = const()[name = string("normed_65_axes_0"), val = tensor([-1])]; + fp16 var_2962_to_fp16 = const()[name = string("op_2962_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_65_cast_fp16 = layer_norm(axes = normed_65_axes_0, epsilon = var_2962_to_fp16, x = input_73_cast_fp16)[name = string("normed_65_cast_fp16")]; + tensor normed_67_begin_0 = const()[name = string("normed_67_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_67_end_0 = const()[name = string("normed_67_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_67_end_mask_0 = const()[name = string("normed_67_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_67_cast_fp16 = slice_by_index(begin = normed_67_begin_0, end = normed_67_end_0, end_mask = normed_67_end_mask_0, x = normed_65_cast_fp16)[name = string("normed_67_cast_fp16")]; + tensor const_140_promoted_to_fp16 = const()[name = string("const_140_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680835264)))]; + tensor hidden_states_43_cast_fp16 = mul(x = normed_67_cast_fp16, y = const_140_promoted_to_fp16)[name = string("hidden_states_43_cast_fp16")]; + tensor var_2990 = const()[name = string("op_2990"), val = tensor([0, 2, 1])]; + tensor var_2993_axes_0 = const()[name = string("op_2993_axes_0"), val = tensor([2])]; + tensor var_2991_cast_fp16 = transpose(perm = var_2990, x = hidden_states_43_cast_fp16)[name = string("transpose_89")]; + tensor var_2993_cast_fp16 = expand_dims(axes = var_2993_axes_0, x = var_2991_cast_fp16)[name = string("op_2993_cast_fp16")]; + string query_states_33_pad_type_0 = const()[name = string("query_states_33_pad_type_0"), val = string("valid")]; + tensor query_states_33_strides_0 = const()[name = string("query_states_33_strides_0"), val = tensor([1, 1])]; + tensor query_states_33_pad_0 = const()[name = string("query_states_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_33_dilations_0 = const()[name = string("query_states_33_dilations_0"), val = tensor([1, 1])]; + int32 query_states_33_groups_0 = const()[name = string("query_states_33_groups_0"), val = int32(1)]; + tensor query_states_33 = conv(dilations = query_states_33_dilations_0, groups = query_states_33_groups_0, pad = query_states_33_pad_0, pad_type = query_states_33_pad_type_0, strides = query_states_33_strides_0, weight = model_model_layers_18_self_attn_q_proj_weight_palettized, x = var_2993_cast_fp16)[name = string("query_states_33")]; + string key_states_41_pad_type_0 = const()[name = string("key_states_41_pad_type_0"), val = string("valid")]; + tensor key_states_41_strides_0 = const()[name = string("key_states_41_strides_0"), val = tensor([1, 1])]; + tensor key_states_41_pad_0 = const()[name = string("key_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_41_dilations_0 = const()[name = string("key_states_41_dilations_0"), val = tensor([1, 1])]; + int32 key_states_41_groups_0 = const()[name = string("key_states_41_groups_0"), val = int32(1)]; + tensor key_states_41 = conv(dilations = key_states_41_dilations_0, groups = key_states_41_groups_0, pad = key_states_41_pad_0, pad_type = key_states_41_pad_type_0, strides = key_states_41_strides_0, weight = model_model_layers_18_self_attn_k_proj_weight_palettized, x = var_2993_cast_fp16)[name = string("key_states_41")]; + string value_states_33_pad_type_0 = const()[name = string("value_states_33_pad_type_0"), val = string("valid")]; + tensor value_states_33_strides_0 = const()[name = string("value_states_33_strides_0"), val = tensor([1, 1])]; + tensor value_states_33_pad_0 = const()[name = string("value_states_33_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_33_dilations_0 = const()[name = string("value_states_33_dilations_0"), val = tensor([1, 1])]; + int32 value_states_33_groups_0 = const()[name = string("value_states_33_groups_0"), val = int32(1)]; + tensor value_states_33 = conv(dilations = value_states_33_dilations_0, groups = value_states_33_groups_0, pad = value_states_33_pad_0, pad_type = value_states_33_pad_type_0, strides = value_states_33_strides_0, weight = model_model_layers_18_self_attn_v_proj_weight_palettized, x = var_2993_cast_fp16)[name = string("value_states_33")]; + tensor var_3035 = const()[name = string("op_3035"), val = tensor([1, 16, 128, 128])]; + tensor var_3036 = reshape(shape = var_3035, x = query_states_33)[name = string("op_3036")]; + tensor var_3041 = const()[name = string("op_3041"), val = tensor([0, 1, 3, 2])]; + tensor var_3046 = const()[name = string("op_3046"), val = tensor([1, 8, 128, 128])]; + tensor var_3047 = reshape(shape = var_3046, x = key_states_41)[name = string("op_3047")]; + tensor var_3052 = const()[name = string("op_3052"), val = tensor([0, 1, 3, 2])]; + tensor var_3057 = const()[name = string("op_3057"), val = tensor([1, 8, 128, 128])]; + tensor var_3058 = reshape(shape = var_3057, x = value_states_33)[name = string("op_3058")]; + tensor var_3063 = const()[name = string("op_3063"), val = tensor([0, 1, 3, 2])]; + int32 var_3074 = const()[name = string("op_3074"), val = int32(-1)]; + fp16 const_142_promoted = const()[name = string("const_142_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_45 = transpose(perm = var_3041, x = var_3036)[name = string("transpose_88")]; + tensor var_3076 = mul(x = hidden_states_45, y = const_142_promoted)[name = string("op_3076")]; + bool input_77_interleave_0 = const()[name = string("input_77_interleave_0"), val = bool(false)]; + tensor input_77 = concat(axis = var_3074, interleave = input_77_interleave_0, values = (hidden_states_45, var_3076))[name = string("input_77")]; + tensor normed_69_axes_0 = const()[name = string("normed_69_axes_0"), val = tensor([-1])]; + fp16 var_3071_to_fp16 = const()[name = string("op_3071_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_69_cast_fp16 = layer_norm(axes = normed_69_axes_0, epsilon = var_3071_to_fp16, x = input_77)[name = string("normed_69_cast_fp16")]; + tensor normed_71_begin_0 = const()[name = string("normed_71_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_71_end_0 = const()[name = string("normed_71_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_71_end_mask_0 = const()[name = string("normed_71_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_71 = slice_by_index(begin = normed_71_begin_0, end = normed_71_end_0, end_mask = normed_71_end_mask_0, x = normed_69_cast_fp16)[name = string("normed_71")]; + tensor const_145 = const()[name = string("const_145"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680839424)))]; + tensor q_9 = mul(x = normed_71, y = const_145)[name = string("q_9")]; + int32 var_3099 = const()[name = string("op_3099"), val = int32(-1)]; + fp16 const_146_promoted = const()[name = string("const_146_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_47 = transpose(perm = var_3052, x = var_3047)[name = string("transpose_87")]; + tensor var_3101 = mul(x = hidden_states_47, y = const_146_promoted)[name = string("op_3101")]; + bool input_79_interleave_0 = const()[name = string("input_79_interleave_0"), val = bool(false)]; + tensor input_79 = concat(axis = var_3099, interleave = input_79_interleave_0, values = (hidden_states_47, var_3101))[name = string("input_79")]; + tensor normed_73_axes_0 = const()[name = string("normed_73_axes_0"), val = tensor([-1])]; + fp16 var_3096_to_fp16 = const()[name = string("op_3096_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_73_cast_fp16 = layer_norm(axes = normed_73_axes_0, epsilon = var_3096_to_fp16, x = input_79)[name = string("normed_73_cast_fp16")]; + tensor normed_75_begin_0 = const()[name = string("normed_75_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_75_end_0 = const()[name = string("normed_75_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_75_end_mask_0 = const()[name = string("normed_75_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_75 = slice_by_index(begin = normed_75_begin_0, end = normed_75_end_0, end_mask = normed_75_end_mask_0, x = normed_73_cast_fp16)[name = string("normed_75")]; + tensor const_149 = const()[name = string("const_149"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680839744)))]; + tensor k_9 = mul(x = normed_75, y = const_149)[name = string("k_9")]; + tensor var_3127 = mul(x = q_9, y = cos_5)[name = string("op_3127")]; + tensor x1_17_begin_0 = const()[name = string("x1_17_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_17_end_0 = const()[name = string("x1_17_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_17_end_mask_0 = const()[name = string("x1_17_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_17 = slice_by_index(begin = x1_17_begin_0, end = x1_17_end_0, end_mask = x1_17_end_mask_0, x = q_9)[name = string("x1_17")]; + tensor x2_17_begin_0 = const()[name = string("x2_17_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_17_end_0 = const()[name = string("x2_17_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_17_end_mask_0 = const()[name = string("x2_17_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_17 = slice_by_index(begin = x2_17_begin_0, end = x2_17_end_0, end_mask = x2_17_end_mask_0, x = q_9)[name = string("x2_17")]; + fp16 const_152_promoted = const()[name = string("const_152_promoted"), val = fp16(-0x1p+0)]; + tensor var_3148 = mul(x = x2_17, y = const_152_promoted)[name = string("op_3148")]; + int32 var_3150 = const()[name = string("op_3150"), val = int32(-1)]; + bool var_3151_interleave_0 = const()[name = string("op_3151_interleave_0"), val = bool(false)]; + tensor var_3151 = concat(axis = var_3150, interleave = var_3151_interleave_0, values = (var_3148, x1_17))[name = string("op_3151")]; + tensor var_3152 = mul(x = var_3151, y = sin_5)[name = string("op_3152")]; + tensor query_states_35 = add(x = var_3127, y = var_3152)[name = string("query_states_35")]; + tensor var_3155 = mul(x = k_9, y = cos_5)[name = string("op_3155")]; + tensor x1_19_begin_0 = const()[name = string("x1_19_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_19_end_0 = const()[name = string("x1_19_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_19_end_mask_0 = const()[name = string("x1_19_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_19 = slice_by_index(begin = x1_19_begin_0, end = x1_19_end_0, end_mask = x1_19_end_mask_0, x = k_9)[name = string("x1_19")]; + tensor x2_19_begin_0 = const()[name = string("x2_19_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_19_end_0 = const()[name = string("x2_19_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_19_end_mask_0 = const()[name = string("x2_19_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_19 = slice_by_index(begin = x2_19_begin_0, end = x2_19_end_0, end_mask = x2_19_end_mask_0, x = k_9)[name = string("x2_19")]; + fp16 const_155_promoted = const()[name = string("const_155_promoted"), val = fp16(-0x1p+0)]; + tensor var_3176 = mul(x = x2_19, y = const_155_promoted)[name = string("op_3176")]; + int32 var_3178 = const()[name = string("op_3178"), val = int32(-1)]; + bool var_3179_interleave_0 = const()[name = string("op_3179_interleave_0"), val = bool(false)]; + tensor var_3179 = concat(axis = var_3178, interleave = var_3179_interleave_0, values = (var_3176, x1_19))[name = string("op_3179")]; + tensor var_3180 = mul(x = var_3179, y = sin_5)[name = string("op_3180")]; + tensor key_states_43 = add(x = var_3155, y = var_3180)[name = string("key_states_43")]; + tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([18])]; + tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; + tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; + tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([19])]; + int32 concat_74_axis_0 = const()[name = string("concat_74_axis_0"), val = int32(0)]; + bool concat_74_interleave_0 = const()[name = string("concat_74_interleave_0"), val = bool(false)]; + tensor concat_74 = concat(axis = concat_74_axis_0, interleave = concat_74_interleave_0, values = (expand_dims_48, expand_dims_49, current_pos, expand_dims_51))[name = string("concat_74")]; + tensor concat_75_values1_0 = const()[name = string("concat_75_values1_0"), val = tensor([0])]; + tensor concat_75_values3_0 = const()[name = string("concat_75_values3_0"), val = tensor([0])]; + int32 concat_75_axis_0 = const()[name = string("concat_75_axis_0"), val = int32(0)]; + bool concat_75_interleave_0 = const()[name = string("concat_75_interleave_0"), val = bool(false)]; + tensor concat_75 = concat(axis = concat_75_axis_0, interleave = concat_75_interleave_0, values = (expand_dims_52, concat_75_values1_0, var_1042, concat_75_values3_0))[name = string("concat_75")]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_74, begin_mask = model_model_kv_cache_0_internal_tensor_assign_9_begin_mask_0, end = concat_75, end_mask = model_model_kv_cache_0_internal_tensor_assign_9_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_9_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_9_stride_0, update = key_states_43, x = coreml_update_state_35)[name = string("model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_9_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_92_write_state")]; + tensor coreml_update_state_36 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_92")]; + tensor expand_dims_54 = const()[name = string("expand_dims_54"), val = tensor([46])]; + tensor expand_dims_55 = const()[name = string("expand_dims_55"), val = tensor([0])]; + tensor expand_dims_57 = const()[name = string("expand_dims_57"), val = tensor([0])]; + tensor expand_dims_58 = const()[name = string("expand_dims_58"), val = tensor([47])]; + int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; + bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; + tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_54, expand_dims_55, current_pos, expand_dims_57))[name = string("concat_78")]; + tensor concat_79_values1_0 = const()[name = string("concat_79_values1_0"), val = tensor([0])]; + tensor concat_79_values3_0 = const()[name = string("concat_79_values3_0"), val = tensor([0])]; + int32 concat_79_axis_0 = const()[name = string("concat_79_axis_0"), val = int32(0)]; + bool concat_79_interleave_0 = const()[name = string("concat_79_interleave_0"), val = bool(false)]; + tensor concat_79 = concat(axis = concat_79_axis_0, interleave = concat_79_interleave_0, values = (expand_dims_58, concat_79_values1_0, var_1042, concat_79_values3_0))[name = string("concat_79")]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_35 = transpose(perm = var_3063, x = var_3058)[name = string("transpose_86")]; + tensor model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_78, begin_mask = model_model_kv_cache_0_internal_tensor_assign_10_begin_mask_0, end = concat_79, end_mask = model_model_kv_cache_0_internal_tensor_assign_10_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_10_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_10_stride_0, update = value_states_35, x = coreml_update_state_36)[name = string("model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_10_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_93_write_state")]; + tensor coreml_update_state_37 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_93")]; + tensor var_3251_begin_0 = const()[name = string("op_3251_begin_0"), val = tensor([18, 0, 0, 0])]; + tensor var_3251_end_0 = const()[name = string("op_3251_end_0"), val = tensor([19, 8, 1024, 128])]; + tensor var_3251_end_mask_0 = const()[name = string("op_3251_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3251_cast_fp16 = slice_by_index(begin = var_3251_begin_0, end = var_3251_end_0, end_mask = var_3251_end_mask_0, x = coreml_update_state_37)[name = string("op_3251_cast_fp16")]; + tensor K_layer_cache_9_axes_0 = const()[name = string("K_layer_cache_9_axes_0"), val = tensor([0])]; + tensor K_layer_cache_9_cast_fp16 = squeeze(axes = K_layer_cache_9_axes_0, x = var_3251_cast_fp16)[name = string("K_layer_cache_9_cast_fp16")]; + tensor var_3258_begin_0 = const()[name = string("op_3258_begin_0"), val = tensor([46, 0, 0, 0])]; + tensor var_3258_end_0 = const()[name = string("op_3258_end_0"), val = tensor([47, 8, 1024, 128])]; + tensor var_3258_end_mask_0 = const()[name = string("op_3258_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3258_cast_fp16 = slice_by_index(begin = var_3258_begin_0, end = var_3258_end_0, end_mask = var_3258_end_mask_0, x = coreml_update_state_37)[name = string("op_3258_cast_fp16")]; + tensor V_layer_cache_9_axes_0 = const()[name = string("V_layer_cache_9_axes_0"), val = tensor([0])]; + tensor V_layer_cache_9_cast_fp16 = squeeze(axes = V_layer_cache_9_axes_0, x = var_3258_cast_fp16)[name = string("V_layer_cache_9_cast_fp16")]; + tensor x_67_axes_0 = const()[name = string("x_67_axes_0"), val = tensor([1])]; + tensor x_67_cast_fp16 = expand_dims(axes = x_67_axes_0, x = K_layer_cache_9_cast_fp16)[name = string("x_67_cast_fp16")]; + tensor var_3287 = const()[name = string("op_3287"), val = tensor([1, 2, 1, 1])]; + tensor x_69_cast_fp16 = tile(reps = var_3287, x = x_67_cast_fp16)[name = string("x_69_cast_fp16")]; + tensor var_3299 = const()[name = string("op_3299"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_47_cast_fp16 = reshape(shape = var_3299, x = x_69_cast_fp16)[name = string("key_states_47_cast_fp16")]; + tensor x_73_axes_0 = const()[name = string("x_73_axes_0"), val = tensor([1])]; + tensor x_73_cast_fp16 = expand_dims(axes = x_73_axes_0, x = V_layer_cache_9_cast_fp16)[name = string("x_73_cast_fp16")]; + tensor var_3307 = const()[name = string("op_3307"), val = tensor([1, 2, 1, 1])]; + tensor x_75_cast_fp16 = tile(reps = var_3307, x = x_73_cast_fp16)[name = string("x_75_cast_fp16")]; + bool var_3334_transpose_x_0 = const()[name = string("op_3334_transpose_x_0"), val = bool(false)]; + bool var_3334_transpose_y_0 = const()[name = string("op_3334_transpose_y_0"), val = bool(true)]; + tensor var_3334 = matmul(transpose_x = var_3334_transpose_x_0, transpose_y = var_3334_transpose_y_0, x = query_states_35, y = key_states_47_cast_fp16)[name = string("op_3334")]; + fp16 var_3335_to_fp16 = const()[name = string("op_3335_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_17_cast_fp16 = mul(x = var_3334, y = var_3335_to_fp16)[name = string("attn_weights_17_cast_fp16")]; + tensor attn_weights_19_cast_fp16 = add(x = attn_weights_17_cast_fp16, y = causal_mask)[name = string("attn_weights_19_cast_fp16")]; + int32 var_3370 = const()[name = string("op_3370"), val = int32(-1)]; + tensor var_3372_cast_fp16 = softmax(axis = var_3370, x = attn_weights_19_cast_fp16)[name = string("op_3372_cast_fp16")]; + tensor concat_84 = const()[name = string("concat_84"), val = tensor([16, 128, 1024])]; + tensor reshape_12_cast_fp16 = reshape(shape = concat_84, x = var_3372_cast_fp16)[name = string("reshape_12_cast_fp16")]; + tensor concat_85 = const()[name = string("concat_85"), val = tensor([16, 1024, 128])]; + tensor reshape_13_cast_fp16 = reshape(shape = concat_85, x = x_75_cast_fp16)[name = string("reshape_13_cast_fp16")]; + bool matmul_4_transpose_x_0 = const()[name = string("matmul_4_transpose_x_0"), val = bool(false)]; + bool matmul_4_transpose_y_0 = const()[name = string("matmul_4_transpose_y_0"), val = bool(false)]; + tensor matmul_4_cast_fp16 = matmul(transpose_x = matmul_4_transpose_x_0, transpose_y = matmul_4_transpose_y_0, x = reshape_12_cast_fp16, y = reshape_13_cast_fp16)[name = string("matmul_4_cast_fp16")]; + tensor concat_89 = const()[name = string("concat_89"), val = tensor([1, 16, 128, 128])]; + tensor reshape_14_cast_fp16 = reshape(shape = concat_89, x = matmul_4_cast_fp16)[name = string("reshape_14_cast_fp16")]; + tensor var_3384_perm_0 = const()[name = string("op_3384_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_3403 = const()[name = string("op_3403"), val = tensor([1, 128, 2048])]; + tensor var_3384_cast_fp16 = transpose(perm = var_3384_perm_0, x = reshape_14_cast_fp16)[name = string("transpose_85")]; + tensor attn_output_45_cast_fp16 = reshape(shape = var_3403, x = var_3384_cast_fp16)[name = string("attn_output_45_cast_fp16")]; + tensor var_3408 = const()[name = string("op_3408"), val = tensor([0, 2, 1])]; + string var_3424_pad_type_0 = const()[name = string("op_3424_pad_type_0"), val = string("valid")]; + int32 var_3424_groups_0 = const()[name = string("op_3424_groups_0"), val = int32(1)]; + tensor var_3424_strides_0 = const()[name = string("op_3424_strides_0"), val = tensor([1])]; + tensor var_3424_pad_0 = const()[name = string("op_3424_pad_0"), val = tensor([0, 0])]; + tensor var_3424_dilations_0 = const()[name = string("op_3424_dilations_0"), val = tensor([1])]; + tensor squeeze_4_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(680840064))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685034432))))[name = string("squeeze_4_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_3409_cast_fp16 = transpose(perm = var_3408, x = attn_output_45_cast_fp16)[name = string("transpose_84")]; + tensor var_3424_cast_fp16 = conv(dilations = var_3424_dilations_0, groups = var_3424_groups_0, pad = var_3424_pad_0, pad_type = var_3424_pad_type_0, strides = var_3424_strides_0, weight = squeeze_4_cast_fp16_to_fp32_to_fp16_palettized, x = var_3409_cast_fp16)[name = string("op_3424_cast_fp16")]; + tensor var_3428 = const()[name = string("op_3428"), val = tensor([0, 2, 1])]; + tensor attn_output_49_cast_fp16 = transpose(perm = var_3428, x = var_3424_cast_fp16)[name = string("transpose_83")]; + tensor hidden_states_49_cast_fp16 = add(x = hidden_states_41_cast_fp16, y = attn_output_49_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; + int32 var_3441 = const()[name = string("op_3441"), val = int32(-1)]; + fp16 const_167_promoted_to_fp16 = const()[name = string("const_167_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3443_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_167_promoted_to_fp16)[name = string("op_3443_cast_fp16")]; + bool input_83_interleave_0 = const()[name = string("input_83_interleave_0"), val = bool(false)]; + tensor input_83_cast_fp16 = concat(axis = var_3441, interleave = input_83_interleave_0, values = (hidden_states_49_cast_fp16, var_3443_cast_fp16))[name = string("input_83_cast_fp16")]; + tensor normed_77_axes_0 = const()[name = string("normed_77_axes_0"), val = tensor([-1])]; + fp16 var_3438_to_fp16 = const()[name = string("op_3438_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_77_cast_fp16 = layer_norm(axes = normed_77_axes_0, epsilon = var_3438_to_fp16, x = input_83_cast_fp16)[name = string("normed_77_cast_fp16")]; + tensor normed_79_begin_0 = const()[name = string("normed_79_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_79_end_0 = const()[name = string("normed_79_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_79_end_mask_0 = const()[name = string("normed_79_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_79_cast_fp16 = slice_by_index(begin = normed_79_begin_0, end = normed_79_end_0, end_mask = normed_79_end_mask_0, x = normed_77_cast_fp16)[name = string("normed_79_cast_fp16")]; + tensor const_170_promoted_to_fp16 = const()[name = string("const_170_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685165568)))]; + tensor x_77_cast_fp16 = mul(x = normed_79_cast_fp16, y = const_170_promoted_to_fp16)[name = string("x_77_cast_fp16")]; + tensor var_3468 = const()[name = string("op_3468"), val = tensor([0, 2, 1])]; + tensor input_85_axes_0 = const()[name = string("input_85_axes_0"), val = tensor([2])]; + tensor var_3469 = transpose(perm = var_3468, x = x_77_cast_fp16)[name = string("transpose_82")]; + tensor input_85 = expand_dims(axes = input_85_axes_0, x = var_3469)[name = string("input_85")]; + string input_87_pad_type_0 = const()[name = string("input_87_pad_type_0"), val = string("valid")]; + tensor input_87_strides_0 = const()[name = string("input_87_strides_0"), val = tensor([1, 1])]; + tensor input_87_pad_0 = const()[name = string("input_87_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_87_dilations_0 = const()[name = string("input_87_dilations_0"), val = tensor([1, 1])]; + int32 input_87_groups_0 = const()[name = string("input_87_groups_0"), val = int32(1)]; + tensor input_87 = conv(dilations = input_87_dilations_0, groups = input_87_groups_0, pad = input_87_pad_0, pad_type = input_87_pad_type_0, strides = input_87_strides_0, weight = model_model_layers_18_mlp_gate_proj_weight_palettized, x = input_85)[name = string("input_87")]; + string b_9_pad_type_0 = const()[name = string("b_9_pad_type_0"), val = string("valid")]; + tensor b_9_strides_0 = const()[name = string("b_9_strides_0"), val = tensor([1, 1])]; + tensor b_9_pad_0 = const()[name = string("b_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_9_dilations_0 = const()[name = string("b_9_dilations_0"), val = tensor([1, 1])]; + int32 b_9_groups_0 = const()[name = string("b_9_groups_0"), val = int32(1)]; + tensor b_9 = conv(dilations = b_9_dilations_0, groups = b_9_groups_0, pad = b_9_pad_0, pad_type = b_9_pad_type_0, strides = b_9_strides_0, weight = model_model_layers_18_mlp_up_proj_weight_palettized, x = input_85)[name = string("b_9")]; + tensor c_9 = silu(x = input_87)[name = string("c_9")]; + tensor input_89 = mul(x = c_9, y = b_9)[name = string("input_89")]; + string e_9_pad_type_0 = const()[name = string("e_9_pad_type_0"), val = string("valid")]; + tensor e_9_strides_0 = const()[name = string("e_9_strides_0"), val = tensor([1, 1])]; + tensor e_9_pad_0 = const()[name = string("e_9_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_9_dilations_0 = const()[name = string("e_9_dilations_0"), val = tensor([1, 1])]; + int32 e_9_groups_0 = const()[name = string("e_9_groups_0"), val = int32(1)]; + tensor e_9 = conv(dilations = e_9_dilations_0, groups = e_9_groups_0, pad = e_9_pad_0, pad_type = e_9_pad_type_0, strides = e_9_strides_0, weight = model_model_layers_18_mlp_down_proj_weight_palettized, x = input_89)[name = string("e_9")]; + tensor var_3491_axes_0 = const()[name = string("op_3491_axes_0"), val = tensor([2])]; + tensor var_3491 = squeeze(axes = var_3491_axes_0, x = e_9)[name = string("op_3491")]; + tensor var_3492 = const()[name = string("op_3492"), val = tensor([0, 2, 1])]; + tensor var_3493 = transpose(perm = var_3492, x = var_3491)[name = string("transpose_81")]; + tensor hidden_states_51_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = var_3493)[name = string("hidden_states_51_cast_fp16")]; + int32 var_3505 = const()[name = string("op_3505"), val = int32(-1)]; + fp16 const_171_promoted_to_fp16 = const()[name = string("const_171_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3507_cast_fp16 = mul(x = hidden_states_51_cast_fp16, y = const_171_promoted_to_fp16)[name = string("op_3507_cast_fp16")]; + bool input_91_interleave_0 = const()[name = string("input_91_interleave_0"), val = bool(false)]; + tensor input_91_cast_fp16 = concat(axis = var_3505, interleave = input_91_interleave_0, values = (hidden_states_51_cast_fp16, var_3507_cast_fp16))[name = string("input_91_cast_fp16")]; + tensor normed_81_axes_0 = const()[name = string("normed_81_axes_0"), val = tensor([-1])]; + fp16 var_3502_to_fp16 = const()[name = string("op_3502_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_81_cast_fp16 = layer_norm(axes = normed_81_axes_0, epsilon = var_3502_to_fp16, x = input_91_cast_fp16)[name = string("normed_81_cast_fp16")]; + tensor normed_83_begin_0 = const()[name = string("normed_83_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_83_end_0 = const()[name = string("normed_83_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_83_end_mask_0 = const()[name = string("normed_83_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_83_cast_fp16 = slice_by_index(begin = normed_83_begin_0, end = normed_83_end_0, end_mask = normed_83_end_mask_0, x = normed_81_cast_fp16)[name = string("normed_83_cast_fp16")]; + tensor const_174_promoted_to_fp16 = const()[name = string("const_174_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685169728)))]; + tensor hidden_states_53_cast_fp16 = mul(x = normed_83_cast_fp16, y = const_174_promoted_to_fp16)[name = string("hidden_states_53_cast_fp16")]; + tensor var_3530 = const()[name = string("op_3530"), val = tensor([0, 2, 1])]; + tensor var_3533_axes_0 = const()[name = string("op_3533_axes_0"), val = tensor([2])]; + tensor var_3531_cast_fp16 = transpose(perm = var_3530, x = hidden_states_53_cast_fp16)[name = string("transpose_80")]; + tensor var_3533_cast_fp16 = expand_dims(axes = var_3533_axes_0, x = var_3531_cast_fp16)[name = string("op_3533_cast_fp16")]; + string query_states_41_pad_type_0 = const()[name = string("query_states_41_pad_type_0"), val = string("valid")]; + tensor query_states_41_strides_0 = const()[name = string("query_states_41_strides_0"), val = tensor([1, 1])]; + tensor query_states_41_pad_0 = const()[name = string("query_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_41_dilations_0 = const()[name = string("query_states_41_dilations_0"), val = tensor([1, 1])]; + int32 query_states_41_groups_0 = const()[name = string("query_states_41_groups_0"), val = int32(1)]; + tensor query_states_41 = conv(dilations = query_states_41_dilations_0, groups = query_states_41_groups_0, pad = query_states_41_pad_0, pad_type = query_states_41_pad_type_0, strides = query_states_41_strides_0, weight = model_model_layers_19_self_attn_q_proj_weight_palettized, x = var_3533_cast_fp16)[name = string("query_states_41")]; + string key_states_51_pad_type_0 = const()[name = string("key_states_51_pad_type_0"), val = string("valid")]; + tensor key_states_51_strides_0 = const()[name = string("key_states_51_strides_0"), val = tensor([1, 1])]; + tensor key_states_51_pad_0 = const()[name = string("key_states_51_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_51_dilations_0 = const()[name = string("key_states_51_dilations_0"), val = tensor([1, 1])]; + int32 key_states_51_groups_0 = const()[name = string("key_states_51_groups_0"), val = int32(1)]; + tensor key_states_51 = conv(dilations = key_states_51_dilations_0, groups = key_states_51_groups_0, pad = key_states_51_pad_0, pad_type = key_states_51_pad_type_0, strides = key_states_51_strides_0, weight = model_model_layers_19_self_attn_k_proj_weight_palettized, x = var_3533_cast_fp16)[name = string("key_states_51")]; + string value_states_41_pad_type_0 = const()[name = string("value_states_41_pad_type_0"), val = string("valid")]; + tensor value_states_41_strides_0 = const()[name = string("value_states_41_strides_0"), val = tensor([1, 1])]; + tensor value_states_41_pad_0 = const()[name = string("value_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_41_dilations_0 = const()[name = string("value_states_41_dilations_0"), val = tensor([1, 1])]; + int32 value_states_41_groups_0 = const()[name = string("value_states_41_groups_0"), val = int32(1)]; + tensor value_states_41 = conv(dilations = value_states_41_dilations_0, groups = value_states_41_groups_0, pad = value_states_41_pad_0, pad_type = value_states_41_pad_type_0, strides = value_states_41_strides_0, weight = model_model_layers_19_self_attn_v_proj_weight_palettized, x = var_3533_cast_fp16)[name = string("value_states_41")]; + tensor var_3575 = const()[name = string("op_3575"), val = tensor([1, 16, 128, 128])]; + tensor var_3576 = reshape(shape = var_3575, x = query_states_41)[name = string("op_3576")]; + tensor var_3581 = const()[name = string("op_3581"), val = tensor([0, 1, 3, 2])]; + tensor var_3586 = const()[name = string("op_3586"), val = tensor([1, 8, 128, 128])]; + tensor var_3587 = reshape(shape = var_3586, x = key_states_51)[name = string("op_3587")]; + tensor var_3592 = const()[name = string("op_3592"), val = tensor([0, 1, 3, 2])]; + tensor var_3597 = const()[name = string("op_3597"), val = tensor([1, 8, 128, 128])]; + tensor var_3598 = reshape(shape = var_3597, x = value_states_41)[name = string("op_3598")]; + tensor var_3603 = const()[name = string("op_3603"), val = tensor([0, 1, 3, 2])]; + int32 var_3614 = const()[name = string("op_3614"), val = int32(-1)]; + fp16 const_176_promoted = const()[name = string("const_176_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_55 = transpose(perm = var_3581, x = var_3576)[name = string("transpose_79")]; + tensor var_3616 = mul(x = hidden_states_55, y = const_176_promoted)[name = string("op_3616")]; + bool input_95_interleave_0 = const()[name = string("input_95_interleave_0"), val = bool(false)]; + tensor input_95 = concat(axis = var_3614, interleave = input_95_interleave_0, values = (hidden_states_55, var_3616))[name = string("input_95")]; + tensor normed_85_axes_0 = const()[name = string("normed_85_axes_0"), val = tensor([-1])]; + fp16 var_3611_to_fp16 = const()[name = string("op_3611_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_85_cast_fp16 = layer_norm(axes = normed_85_axes_0, epsilon = var_3611_to_fp16, x = input_95)[name = string("normed_85_cast_fp16")]; + tensor normed_87_begin_0 = const()[name = string("normed_87_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_87_end_0 = const()[name = string("normed_87_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_87_end_mask_0 = const()[name = string("normed_87_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_87 = slice_by_index(begin = normed_87_begin_0, end = normed_87_end_0, end_mask = normed_87_end_mask_0, x = normed_85_cast_fp16)[name = string("normed_87")]; + tensor const_179 = const()[name = string("const_179"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685173888)))]; + tensor q_11 = mul(x = normed_87, y = const_179)[name = string("q_11")]; + int32 var_3639 = const()[name = string("op_3639"), val = int32(-1)]; + fp16 const_180_promoted = const()[name = string("const_180_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_57 = transpose(perm = var_3592, x = var_3587)[name = string("transpose_78")]; + tensor var_3641 = mul(x = hidden_states_57, y = const_180_promoted)[name = string("op_3641")]; + bool input_97_interleave_0 = const()[name = string("input_97_interleave_0"), val = bool(false)]; + tensor input_97 = concat(axis = var_3639, interleave = input_97_interleave_0, values = (hidden_states_57, var_3641))[name = string("input_97")]; + tensor normed_89_axes_0 = const()[name = string("normed_89_axes_0"), val = tensor([-1])]; + fp16 var_3636_to_fp16 = const()[name = string("op_3636_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_89_cast_fp16 = layer_norm(axes = normed_89_axes_0, epsilon = var_3636_to_fp16, x = input_97)[name = string("normed_89_cast_fp16")]; + tensor normed_91_begin_0 = const()[name = string("normed_91_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_91_end_0 = const()[name = string("normed_91_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_91_end_mask_0 = const()[name = string("normed_91_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_91 = slice_by_index(begin = normed_91_begin_0, end = normed_91_end_0, end_mask = normed_91_end_mask_0, x = normed_89_cast_fp16)[name = string("normed_91")]; + tensor const_183 = const()[name = string("const_183"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685174208)))]; + tensor k_11 = mul(x = normed_91, y = const_183)[name = string("k_11")]; + tensor var_3667 = mul(x = q_11, y = cos_5)[name = string("op_3667")]; + tensor x1_21_begin_0 = const()[name = string("x1_21_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_21_end_0 = const()[name = string("x1_21_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_21_end_mask_0 = const()[name = string("x1_21_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_21 = slice_by_index(begin = x1_21_begin_0, end = x1_21_end_0, end_mask = x1_21_end_mask_0, x = q_11)[name = string("x1_21")]; + tensor x2_21_begin_0 = const()[name = string("x2_21_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_21_end_0 = const()[name = string("x2_21_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_21_end_mask_0 = const()[name = string("x2_21_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_21 = slice_by_index(begin = x2_21_begin_0, end = x2_21_end_0, end_mask = x2_21_end_mask_0, x = q_11)[name = string("x2_21")]; + fp16 const_186_promoted = const()[name = string("const_186_promoted"), val = fp16(-0x1p+0)]; + tensor var_3688 = mul(x = x2_21, y = const_186_promoted)[name = string("op_3688")]; + int32 var_3690 = const()[name = string("op_3690"), val = int32(-1)]; + bool var_3691_interleave_0 = const()[name = string("op_3691_interleave_0"), val = bool(false)]; + tensor var_3691 = concat(axis = var_3690, interleave = var_3691_interleave_0, values = (var_3688, x1_21))[name = string("op_3691")]; + tensor var_3692 = mul(x = var_3691, y = sin_5)[name = string("op_3692")]; + tensor query_states_43 = add(x = var_3667, y = var_3692)[name = string("query_states_43")]; + tensor var_3695 = mul(x = k_11, y = cos_5)[name = string("op_3695")]; + tensor x1_23_begin_0 = const()[name = string("x1_23_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_23_end_0 = const()[name = string("x1_23_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_23_end_mask_0 = const()[name = string("x1_23_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_23 = slice_by_index(begin = x1_23_begin_0, end = x1_23_end_0, end_mask = x1_23_end_mask_0, x = k_11)[name = string("x1_23")]; + tensor x2_23_begin_0 = const()[name = string("x2_23_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_23_end_0 = const()[name = string("x2_23_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_23_end_mask_0 = const()[name = string("x2_23_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_23 = slice_by_index(begin = x2_23_begin_0, end = x2_23_end_0, end_mask = x2_23_end_mask_0, x = k_11)[name = string("x2_23")]; + fp16 const_189_promoted = const()[name = string("const_189_promoted"), val = fp16(-0x1p+0)]; + tensor var_3716 = mul(x = x2_23, y = const_189_promoted)[name = string("op_3716")]; + int32 var_3718 = const()[name = string("op_3718"), val = int32(-1)]; + bool var_3719_interleave_0 = const()[name = string("op_3719_interleave_0"), val = bool(false)]; + tensor var_3719 = concat(axis = var_3718, interleave = var_3719_interleave_0, values = (var_3716, x1_23))[name = string("op_3719")]; + tensor var_3720 = mul(x = var_3719, y = sin_5)[name = string("op_3720")]; + tensor key_states_53 = add(x = var_3695, y = var_3720)[name = string("key_states_53")]; + tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([19])]; + tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; + tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; + tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([20])]; + int32 concat_92_axis_0 = const()[name = string("concat_92_axis_0"), val = int32(0)]; + bool concat_92_interleave_0 = const()[name = string("concat_92_interleave_0"), val = bool(false)]; + tensor concat_92 = concat(axis = concat_92_axis_0, interleave = concat_92_interleave_0, values = (expand_dims_60, expand_dims_61, current_pos, expand_dims_63))[name = string("concat_92")]; + tensor concat_93_values1_0 = const()[name = string("concat_93_values1_0"), val = tensor([0])]; + tensor concat_93_values3_0 = const()[name = string("concat_93_values3_0"), val = tensor([0])]; + int32 concat_93_axis_0 = const()[name = string("concat_93_axis_0"), val = int32(0)]; + bool concat_93_interleave_0 = const()[name = string("concat_93_interleave_0"), val = bool(false)]; + tensor concat_93 = concat(axis = concat_93_axis_0, interleave = concat_93_interleave_0, values = (expand_dims_64, concat_93_values1_0, var_1042, concat_93_values3_0))[name = string("concat_93")]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_92, begin_mask = model_model_kv_cache_0_internal_tensor_assign_11_begin_mask_0, end = concat_93, end_mask = model_model_kv_cache_0_internal_tensor_assign_11_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_11_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_11_stride_0, update = key_states_53, x = coreml_update_state_37)[name = string("model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_11_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_94_write_state")]; + tensor coreml_update_state_38 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_94")]; + tensor expand_dims_66 = const()[name = string("expand_dims_66"), val = tensor([47])]; + tensor expand_dims_67 = const()[name = string("expand_dims_67"), val = tensor([0])]; + tensor expand_dims_69 = const()[name = string("expand_dims_69"), val = tensor([0])]; + tensor expand_dims_70 = const()[name = string("expand_dims_70"), val = tensor([48])]; + int32 concat_96_axis_0 = const()[name = string("concat_96_axis_0"), val = int32(0)]; + bool concat_96_interleave_0 = const()[name = string("concat_96_interleave_0"), val = bool(false)]; + tensor concat_96 = concat(axis = concat_96_axis_0, interleave = concat_96_interleave_0, values = (expand_dims_66, expand_dims_67, current_pos, expand_dims_69))[name = string("concat_96")]; + tensor concat_97_values1_0 = const()[name = string("concat_97_values1_0"), val = tensor([0])]; + tensor concat_97_values3_0 = const()[name = string("concat_97_values3_0"), val = tensor([0])]; + int32 concat_97_axis_0 = const()[name = string("concat_97_axis_0"), val = int32(0)]; + bool concat_97_interleave_0 = const()[name = string("concat_97_interleave_0"), val = bool(false)]; + tensor concat_97 = concat(axis = concat_97_axis_0, interleave = concat_97_interleave_0, values = (expand_dims_70, concat_97_values1_0, var_1042, concat_97_values3_0))[name = string("concat_97")]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_43 = transpose(perm = var_3603, x = var_3598)[name = string("transpose_77")]; + tensor model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_96, begin_mask = model_model_kv_cache_0_internal_tensor_assign_12_begin_mask_0, end = concat_97, end_mask = model_model_kv_cache_0_internal_tensor_assign_12_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_12_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_12_stride_0, update = value_states_43, x = coreml_update_state_38)[name = string("model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_12_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_95_write_state")]; + tensor coreml_update_state_39 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_95")]; + tensor var_3791_begin_0 = const()[name = string("op_3791_begin_0"), val = tensor([19, 0, 0, 0])]; + tensor var_3791_end_0 = const()[name = string("op_3791_end_0"), val = tensor([20, 8, 1024, 128])]; + tensor var_3791_end_mask_0 = const()[name = string("op_3791_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3791_cast_fp16 = slice_by_index(begin = var_3791_begin_0, end = var_3791_end_0, end_mask = var_3791_end_mask_0, x = coreml_update_state_39)[name = string("op_3791_cast_fp16")]; + tensor K_layer_cache_11_axes_0 = const()[name = string("K_layer_cache_11_axes_0"), val = tensor([0])]; + tensor K_layer_cache_11_cast_fp16 = squeeze(axes = K_layer_cache_11_axes_0, x = var_3791_cast_fp16)[name = string("K_layer_cache_11_cast_fp16")]; + tensor var_3798_begin_0 = const()[name = string("op_3798_begin_0"), val = tensor([47, 0, 0, 0])]; + tensor var_3798_end_0 = const()[name = string("op_3798_end_0"), val = tensor([48, 8, 1024, 128])]; + tensor var_3798_end_mask_0 = const()[name = string("op_3798_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_3798_cast_fp16 = slice_by_index(begin = var_3798_begin_0, end = var_3798_end_0, end_mask = var_3798_end_mask_0, x = coreml_update_state_39)[name = string("op_3798_cast_fp16")]; + tensor V_layer_cache_11_axes_0 = const()[name = string("V_layer_cache_11_axes_0"), val = tensor([0])]; + tensor V_layer_cache_11_cast_fp16 = squeeze(axes = V_layer_cache_11_axes_0, x = var_3798_cast_fp16)[name = string("V_layer_cache_11_cast_fp16")]; + tensor x_83_axes_0 = const()[name = string("x_83_axes_0"), val = tensor([1])]; + tensor x_83_cast_fp16 = expand_dims(axes = x_83_axes_0, x = K_layer_cache_11_cast_fp16)[name = string("x_83_cast_fp16")]; + tensor var_3827 = const()[name = string("op_3827"), val = tensor([1, 2, 1, 1])]; + tensor x_85_cast_fp16 = tile(reps = var_3827, x = x_83_cast_fp16)[name = string("x_85_cast_fp16")]; + tensor var_3839 = const()[name = string("op_3839"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_57_cast_fp16 = reshape(shape = var_3839, x = x_85_cast_fp16)[name = string("key_states_57_cast_fp16")]; + tensor x_89_axes_0 = const()[name = string("x_89_axes_0"), val = tensor([1])]; + tensor x_89_cast_fp16 = expand_dims(axes = x_89_axes_0, x = V_layer_cache_11_cast_fp16)[name = string("x_89_cast_fp16")]; + tensor var_3847 = const()[name = string("op_3847"), val = tensor([1, 2, 1, 1])]; + tensor x_91_cast_fp16 = tile(reps = var_3847, x = x_89_cast_fp16)[name = string("x_91_cast_fp16")]; + bool var_3874_transpose_x_0 = const()[name = string("op_3874_transpose_x_0"), val = bool(false)]; + bool var_3874_transpose_y_0 = const()[name = string("op_3874_transpose_y_0"), val = bool(true)]; + tensor var_3874 = matmul(transpose_x = var_3874_transpose_x_0, transpose_y = var_3874_transpose_y_0, x = query_states_43, y = key_states_57_cast_fp16)[name = string("op_3874")]; + fp16 var_3875_to_fp16 = const()[name = string("op_3875_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_21_cast_fp16 = mul(x = var_3874, y = var_3875_to_fp16)[name = string("attn_weights_21_cast_fp16")]; + tensor attn_weights_23_cast_fp16 = add(x = attn_weights_21_cast_fp16, y = causal_mask)[name = string("attn_weights_23_cast_fp16")]; + int32 var_3910 = const()[name = string("op_3910"), val = int32(-1)]; + tensor var_3912_cast_fp16 = softmax(axis = var_3910, x = attn_weights_23_cast_fp16)[name = string("op_3912_cast_fp16")]; + tensor concat_102 = const()[name = string("concat_102"), val = tensor([16, 128, 1024])]; + tensor reshape_15_cast_fp16 = reshape(shape = concat_102, x = var_3912_cast_fp16)[name = string("reshape_15_cast_fp16")]; + tensor concat_103 = const()[name = string("concat_103"), val = tensor([16, 1024, 128])]; + tensor reshape_16_cast_fp16 = reshape(shape = concat_103, x = x_91_cast_fp16)[name = string("reshape_16_cast_fp16")]; + bool matmul_5_transpose_x_0 = const()[name = string("matmul_5_transpose_x_0"), val = bool(false)]; + bool matmul_5_transpose_y_0 = const()[name = string("matmul_5_transpose_y_0"), val = bool(false)]; + tensor matmul_5_cast_fp16 = matmul(transpose_x = matmul_5_transpose_x_0, transpose_y = matmul_5_transpose_y_0, x = reshape_15_cast_fp16, y = reshape_16_cast_fp16)[name = string("matmul_5_cast_fp16")]; + tensor concat_107 = const()[name = string("concat_107"), val = tensor([1, 16, 128, 128])]; + tensor reshape_17_cast_fp16 = reshape(shape = concat_107, x = matmul_5_cast_fp16)[name = string("reshape_17_cast_fp16")]; + tensor var_3924_perm_0 = const()[name = string("op_3924_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_3943 = const()[name = string("op_3943"), val = tensor([1, 128, 2048])]; + tensor var_3924_cast_fp16 = transpose(perm = var_3924_perm_0, x = reshape_17_cast_fp16)[name = string("transpose_76")]; + tensor attn_output_55_cast_fp16 = reshape(shape = var_3943, x = var_3924_cast_fp16)[name = string("attn_output_55_cast_fp16")]; + tensor var_3948 = const()[name = string("op_3948"), val = tensor([0, 2, 1])]; + string var_3964_pad_type_0 = const()[name = string("op_3964_pad_type_0"), val = string("valid")]; + int32 var_3964_groups_0 = const()[name = string("op_3964_groups_0"), val = int32(1)]; + tensor var_3964_strides_0 = const()[name = string("op_3964_strides_0"), val = tensor([1])]; + tensor var_3964_pad_0 = const()[name = string("op_3964_pad_0"), val = tensor([0, 0])]; + tensor var_3964_dilations_0 = const()[name = string("op_3964_dilations_0"), val = tensor([1])]; + tensor squeeze_5_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685174528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689368896))))[name = string("squeeze_5_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_3949_cast_fp16 = transpose(perm = var_3948, x = attn_output_55_cast_fp16)[name = string("transpose_75")]; + tensor var_3964_cast_fp16 = conv(dilations = var_3964_dilations_0, groups = var_3964_groups_0, pad = var_3964_pad_0, pad_type = var_3964_pad_type_0, strides = var_3964_strides_0, weight = squeeze_5_cast_fp16_to_fp32_to_fp16_palettized, x = var_3949_cast_fp16)[name = string("op_3964_cast_fp16")]; + tensor var_3968 = const()[name = string("op_3968"), val = tensor([0, 2, 1])]; + tensor attn_output_59_cast_fp16 = transpose(perm = var_3968, x = var_3964_cast_fp16)[name = string("transpose_74")]; + tensor hidden_states_59_cast_fp16 = add(x = hidden_states_51_cast_fp16, y = attn_output_59_cast_fp16)[name = string("hidden_states_59_cast_fp16")]; + int32 var_3981 = const()[name = string("op_3981"), val = int32(-1)]; + fp16 const_201_promoted_to_fp16 = const()[name = string("const_201_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_3983_cast_fp16 = mul(x = hidden_states_59_cast_fp16, y = const_201_promoted_to_fp16)[name = string("op_3983_cast_fp16")]; + bool input_101_interleave_0 = const()[name = string("input_101_interleave_0"), val = bool(false)]; + tensor input_101_cast_fp16 = concat(axis = var_3981, interleave = input_101_interleave_0, values = (hidden_states_59_cast_fp16, var_3983_cast_fp16))[name = string("input_101_cast_fp16")]; + tensor normed_93_axes_0 = const()[name = string("normed_93_axes_0"), val = tensor([-1])]; + fp16 var_3978_to_fp16 = const()[name = string("op_3978_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_93_cast_fp16 = layer_norm(axes = normed_93_axes_0, epsilon = var_3978_to_fp16, x = input_101_cast_fp16)[name = string("normed_93_cast_fp16")]; + tensor normed_95_begin_0 = const()[name = string("normed_95_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_95_end_0 = const()[name = string("normed_95_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_95_end_mask_0 = const()[name = string("normed_95_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_95_cast_fp16 = slice_by_index(begin = normed_95_begin_0, end = normed_95_end_0, end_mask = normed_95_end_mask_0, x = normed_93_cast_fp16)[name = string("normed_95_cast_fp16")]; + tensor const_204_promoted_to_fp16 = const()[name = string("const_204_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689500032)))]; + tensor x_93_cast_fp16 = mul(x = normed_95_cast_fp16, y = const_204_promoted_to_fp16)[name = string("x_93_cast_fp16")]; + tensor var_4008 = const()[name = string("op_4008"), val = tensor([0, 2, 1])]; + tensor input_103_axes_0 = const()[name = string("input_103_axes_0"), val = tensor([2])]; + tensor var_4009 = transpose(perm = var_4008, x = x_93_cast_fp16)[name = string("transpose_73")]; + tensor input_103 = expand_dims(axes = input_103_axes_0, x = var_4009)[name = string("input_103")]; + string input_105_pad_type_0 = const()[name = string("input_105_pad_type_0"), val = string("valid")]; + tensor input_105_strides_0 = const()[name = string("input_105_strides_0"), val = tensor([1, 1])]; + tensor input_105_pad_0 = const()[name = string("input_105_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_105_dilations_0 = const()[name = string("input_105_dilations_0"), val = tensor([1, 1])]; + int32 input_105_groups_0 = const()[name = string("input_105_groups_0"), val = int32(1)]; + tensor input_105 = conv(dilations = input_105_dilations_0, groups = input_105_groups_0, pad = input_105_pad_0, pad_type = input_105_pad_type_0, strides = input_105_strides_0, weight = model_model_layers_19_mlp_gate_proj_weight_palettized, x = input_103)[name = string("input_105")]; + string b_11_pad_type_0 = const()[name = string("b_11_pad_type_0"), val = string("valid")]; + tensor b_11_strides_0 = const()[name = string("b_11_strides_0"), val = tensor([1, 1])]; + tensor b_11_pad_0 = const()[name = string("b_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_11_dilations_0 = const()[name = string("b_11_dilations_0"), val = tensor([1, 1])]; + int32 b_11_groups_0 = const()[name = string("b_11_groups_0"), val = int32(1)]; + tensor b_11 = conv(dilations = b_11_dilations_0, groups = b_11_groups_0, pad = b_11_pad_0, pad_type = b_11_pad_type_0, strides = b_11_strides_0, weight = model_model_layers_19_mlp_up_proj_weight_palettized, x = input_103)[name = string("b_11")]; + tensor c_11 = silu(x = input_105)[name = string("c_11")]; + tensor input_107 = mul(x = c_11, y = b_11)[name = string("input_107")]; + string e_11_pad_type_0 = const()[name = string("e_11_pad_type_0"), val = string("valid")]; + tensor e_11_strides_0 = const()[name = string("e_11_strides_0"), val = tensor([1, 1])]; + tensor e_11_pad_0 = const()[name = string("e_11_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_11_dilations_0 = const()[name = string("e_11_dilations_0"), val = tensor([1, 1])]; + int32 e_11_groups_0 = const()[name = string("e_11_groups_0"), val = int32(1)]; + tensor e_11 = conv(dilations = e_11_dilations_0, groups = e_11_groups_0, pad = e_11_pad_0, pad_type = e_11_pad_type_0, strides = e_11_strides_0, weight = model_model_layers_19_mlp_down_proj_weight_palettized, x = input_107)[name = string("e_11")]; + tensor var_4031_axes_0 = const()[name = string("op_4031_axes_0"), val = tensor([2])]; + tensor var_4031 = squeeze(axes = var_4031_axes_0, x = e_11)[name = string("op_4031")]; + tensor var_4032 = const()[name = string("op_4032"), val = tensor([0, 2, 1])]; + tensor var_4033 = transpose(perm = var_4032, x = var_4031)[name = string("transpose_72")]; + tensor hidden_states_61_cast_fp16 = add(x = hidden_states_59_cast_fp16, y = var_4033)[name = string("hidden_states_61_cast_fp16")]; + int32 var_4045 = const()[name = string("op_4045"), val = int32(-1)]; + fp16 const_205_promoted_to_fp16 = const()[name = string("const_205_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4047_cast_fp16 = mul(x = hidden_states_61_cast_fp16, y = const_205_promoted_to_fp16)[name = string("op_4047_cast_fp16")]; + bool input_109_interleave_0 = const()[name = string("input_109_interleave_0"), val = bool(false)]; + tensor input_109_cast_fp16 = concat(axis = var_4045, interleave = input_109_interleave_0, values = (hidden_states_61_cast_fp16, var_4047_cast_fp16))[name = string("input_109_cast_fp16")]; + tensor normed_97_axes_0 = const()[name = string("normed_97_axes_0"), val = tensor([-1])]; + fp16 var_4042_to_fp16 = const()[name = string("op_4042_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_97_cast_fp16 = layer_norm(axes = normed_97_axes_0, epsilon = var_4042_to_fp16, x = input_109_cast_fp16)[name = string("normed_97_cast_fp16")]; + tensor normed_99_begin_0 = const()[name = string("normed_99_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_99_end_0 = const()[name = string("normed_99_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_99_end_mask_0 = const()[name = string("normed_99_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_99_cast_fp16 = slice_by_index(begin = normed_99_begin_0, end = normed_99_end_0, end_mask = normed_99_end_mask_0, x = normed_97_cast_fp16)[name = string("normed_99_cast_fp16")]; + tensor const_208_promoted_to_fp16 = const()[name = string("const_208_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689504192)))]; + tensor hidden_states_63_cast_fp16 = mul(x = normed_99_cast_fp16, y = const_208_promoted_to_fp16)[name = string("hidden_states_63_cast_fp16")]; + tensor var_4070 = const()[name = string("op_4070"), val = tensor([0, 2, 1])]; + tensor var_4073_axes_0 = const()[name = string("op_4073_axes_0"), val = tensor([2])]; + tensor var_4071_cast_fp16 = transpose(perm = var_4070, x = hidden_states_63_cast_fp16)[name = string("transpose_71")]; + tensor var_4073_cast_fp16 = expand_dims(axes = var_4073_axes_0, x = var_4071_cast_fp16)[name = string("op_4073_cast_fp16")]; + string query_states_49_pad_type_0 = const()[name = string("query_states_49_pad_type_0"), val = string("valid")]; + tensor query_states_49_strides_0 = const()[name = string("query_states_49_strides_0"), val = tensor([1, 1])]; + tensor query_states_49_pad_0 = const()[name = string("query_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_49_dilations_0 = const()[name = string("query_states_49_dilations_0"), val = tensor([1, 1])]; + int32 query_states_49_groups_0 = const()[name = string("query_states_49_groups_0"), val = int32(1)]; + tensor query_states_49 = conv(dilations = query_states_49_dilations_0, groups = query_states_49_groups_0, pad = query_states_49_pad_0, pad_type = query_states_49_pad_type_0, strides = query_states_49_strides_0, weight = model_model_layers_20_self_attn_q_proj_weight_palettized, x = var_4073_cast_fp16)[name = string("query_states_49")]; + string key_states_61_pad_type_0 = const()[name = string("key_states_61_pad_type_0"), val = string("valid")]; + tensor key_states_61_strides_0 = const()[name = string("key_states_61_strides_0"), val = tensor([1, 1])]; + tensor key_states_61_pad_0 = const()[name = string("key_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_61_dilations_0 = const()[name = string("key_states_61_dilations_0"), val = tensor([1, 1])]; + int32 key_states_61_groups_0 = const()[name = string("key_states_61_groups_0"), val = int32(1)]; + tensor key_states_61 = conv(dilations = key_states_61_dilations_0, groups = key_states_61_groups_0, pad = key_states_61_pad_0, pad_type = key_states_61_pad_type_0, strides = key_states_61_strides_0, weight = model_model_layers_20_self_attn_k_proj_weight_palettized, x = var_4073_cast_fp16)[name = string("key_states_61")]; + string value_states_49_pad_type_0 = const()[name = string("value_states_49_pad_type_0"), val = string("valid")]; + tensor value_states_49_strides_0 = const()[name = string("value_states_49_strides_0"), val = tensor([1, 1])]; + tensor value_states_49_pad_0 = const()[name = string("value_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_49_dilations_0 = const()[name = string("value_states_49_dilations_0"), val = tensor([1, 1])]; + int32 value_states_49_groups_0 = const()[name = string("value_states_49_groups_0"), val = int32(1)]; + tensor value_states_49 = conv(dilations = value_states_49_dilations_0, groups = value_states_49_groups_0, pad = value_states_49_pad_0, pad_type = value_states_49_pad_type_0, strides = value_states_49_strides_0, weight = model_model_layers_20_self_attn_v_proj_weight_palettized, x = var_4073_cast_fp16)[name = string("value_states_49")]; + tensor var_4115 = const()[name = string("op_4115"), val = tensor([1, 16, 128, 128])]; + tensor var_4116 = reshape(shape = var_4115, x = query_states_49)[name = string("op_4116")]; + tensor var_4121 = const()[name = string("op_4121"), val = tensor([0, 1, 3, 2])]; + tensor var_4126 = const()[name = string("op_4126"), val = tensor([1, 8, 128, 128])]; + tensor var_4127 = reshape(shape = var_4126, x = key_states_61)[name = string("op_4127")]; + tensor var_4132 = const()[name = string("op_4132"), val = tensor([0, 1, 3, 2])]; + tensor var_4137 = const()[name = string("op_4137"), val = tensor([1, 8, 128, 128])]; + tensor var_4138 = reshape(shape = var_4137, x = value_states_49)[name = string("op_4138")]; + tensor var_4143 = const()[name = string("op_4143"), val = tensor([0, 1, 3, 2])]; + int32 var_4154 = const()[name = string("op_4154"), val = int32(-1)]; + fp16 const_210_promoted = const()[name = string("const_210_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_65 = transpose(perm = var_4121, x = var_4116)[name = string("transpose_70")]; + tensor var_4156 = mul(x = hidden_states_65, y = const_210_promoted)[name = string("op_4156")]; + bool input_113_interleave_0 = const()[name = string("input_113_interleave_0"), val = bool(false)]; + tensor input_113 = concat(axis = var_4154, interleave = input_113_interleave_0, values = (hidden_states_65, var_4156))[name = string("input_113")]; + tensor normed_101_axes_0 = const()[name = string("normed_101_axes_0"), val = tensor([-1])]; + fp16 var_4151_to_fp16 = const()[name = string("op_4151_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_101_cast_fp16 = layer_norm(axes = normed_101_axes_0, epsilon = var_4151_to_fp16, x = input_113)[name = string("normed_101_cast_fp16")]; + tensor normed_103_begin_0 = const()[name = string("normed_103_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_103_end_0 = const()[name = string("normed_103_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_103_end_mask_0 = const()[name = string("normed_103_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_103 = slice_by_index(begin = normed_103_begin_0, end = normed_103_end_0, end_mask = normed_103_end_mask_0, x = normed_101_cast_fp16)[name = string("normed_103")]; + tensor const_213 = const()[name = string("const_213"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689508352)))]; + tensor q_13 = mul(x = normed_103, y = const_213)[name = string("q_13")]; + int32 var_4179 = const()[name = string("op_4179"), val = int32(-1)]; + fp16 const_214_promoted = const()[name = string("const_214_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_67 = transpose(perm = var_4132, x = var_4127)[name = string("transpose_69")]; + tensor var_4181 = mul(x = hidden_states_67, y = const_214_promoted)[name = string("op_4181")]; + bool input_115_interleave_0 = const()[name = string("input_115_interleave_0"), val = bool(false)]; + tensor input_115 = concat(axis = var_4179, interleave = input_115_interleave_0, values = (hidden_states_67, var_4181))[name = string("input_115")]; + tensor normed_105_axes_0 = const()[name = string("normed_105_axes_0"), val = tensor([-1])]; + fp16 var_4176_to_fp16 = const()[name = string("op_4176_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_105_cast_fp16 = layer_norm(axes = normed_105_axes_0, epsilon = var_4176_to_fp16, x = input_115)[name = string("normed_105_cast_fp16")]; + tensor normed_107_begin_0 = const()[name = string("normed_107_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_107_end_0 = const()[name = string("normed_107_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_107_end_mask_0 = const()[name = string("normed_107_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_107 = slice_by_index(begin = normed_107_begin_0, end = normed_107_end_0, end_mask = normed_107_end_mask_0, x = normed_105_cast_fp16)[name = string("normed_107")]; + tensor const_217 = const()[name = string("const_217"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689508672)))]; + tensor k_13 = mul(x = normed_107, y = const_217)[name = string("k_13")]; + tensor var_4207 = mul(x = q_13, y = cos_5)[name = string("op_4207")]; + tensor x1_25_begin_0 = const()[name = string("x1_25_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_25_end_0 = const()[name = string("x1_25_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_25_end_mask_0 = const()[name = string("x1_25_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_25 = slice_by_index(begin = x1_25_begin_0, end = x1_25_end_0, end_mask = x1_25_end_mask_0, x = q_13)[name = string("x1_25")]; + tensor x2_25_begin_0 = const()[name = string("x2_25_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_25_end_0 = const()[name = string("x2_25_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_25_end_mask_0 = const()[name = string("x2_25_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_25 = slice_by_index(begin = x2_25_begin_0, end = x2_25_end_0, end_mask = x2_25_end_mask_0, x = q_13)[name = string("x2_25")]; + fp16 const_220_promoted = const()[name = string("const_220_promoted"), val = fp16(-0x1p+0)]; + tensor var_4228 = mul(x = x2_25, y = const_220_promoted)[name = string("op_4228")]; + int32 var_4230 = const()[name = string("op_4230"), val = int32(-1)]; + bool var_4231_interleave_0 = const()[name = string("op_4231_interleave_0"), val = bool(false)]; + tensor var_4231 = concat(axis = var_4230, interleave = var_4231_interleave_0, values = (var_4228, x1_25))[name = string("op_4231")]; + tensor var_4232 = mul(x = var_4231, y = sin_5)[name = string("op_4232")]; + tensor query_states_51 = add(x = var_4207, y = var_4232)[name = string("query_states_51")]; + tensor var_4235 = mul(x = k_13, y = cos_5)[name = string("op_4235")]; + tensor x1_27_begin_0 = const()[name = string("x1_27_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_27_end_0 = const()[name = string("x1_27_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_27_end_mask_0 = const()[name = string("x1_27_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_27 = slice_by_index(begin = x1_27_begin_0, end = x1_27_end_0, end_mask = x1_27_end_mask_0, x = k_13)[name = string("x1_27")]; + tensor x2_27_begin_0 = const()[name = string("x2_27_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_27_end_0 = const()[name = string("x2_27_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_27_end_mask_0 = const()[name = string("x2_27_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_27 = slice_by_index(begin = x2_27_begin_0, end = x2_27_end_0, end_mask = x2_27_end_mask_0, x = k_13)[name = string("x2_27")]; + fp16 const_223_promoted = const()[name = string("const_223_promoted"), val = fp16(-0x1p+0)]; + tensor var_4256 = mul(x = x2_27, y = const_223_promoted)[name = string("op_4256")]; + int32 var_4258 = const()[name = string("op_4258"), val = int32(-1)]; + bool var_4259_interleave_0 = const()[name = string("op_4259_interleave_0"), val = bool(false)]; + tensor var_4259 = concat(axis = var_4258, interleave = var_4259_interleave_0, values = (var_4256, x1_27))[name = string("op_4259")]; + tensor var_4260 = mul(x = var_4259, y = sin_5)[name = string("op_4260")]; + tensor key_states_63 = add(x = var_4235, y = var_4260)[name = string("key_states_63")]; + tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([20])]; + tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; + tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; + tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([21])]; + int32 concat_110_axis_0 = const()[name = string("concat_110_axis_0"), val = int32(0)]; + bool concat_110_interleave_0 = const()[name = string("concat_110_interleave_0"), val = bool(false)]; + tensor concat_110 = concat(axis = concat_110_axis_0, interleave = concat_110_interleave_0, values = (expand_dims_72, expand_dims_73, current_pos, expand_dims_75))[name = string("concat_110")]; + tensor concat_111_values1_0 = const()[name = string("concat_111_values1_0"), val = tensor([0])]; + tensor concat_111_values3_0 = const()[name = string("concat_111_values3_0"), val = tensor([0])]; + int32 concat_111_axis_0 = const()[name = string("concat_111_axis_0"), val = int32(0)]; + bool concat_111_interleave_0 = const()[name = string("concat_111_interleave_0"), val = bool(false)]; + tensor concat_111 = concat(axis = concat_111_axis_0, interleave = concat_111_interleave_0, values = (expand_dims_76, concat_111_values1_0, var_1042, concat_111_values3_0))[name = string("concat_111")]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_110, begin_mask = model_model_kv_cache_0_internal_tensor_assign_13_begin_mask_0, end = concat_111, end_mask = model_model_kv_cache_0_internal_tensor_assign_13_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_13_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_13_stride_0, update = key_states_63, x = coreml_update_state_39)[name = string("model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_13_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_96_write_state")]; + tensor coreml_update_state_40 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_96")]; + tensor expand_dims_78 = const()[name = string("expand_dims_78"), val = tensor([48])]; + tensor expand_dims_79 = const()[name = string("expand_dims_79"), val = tensor([0])]; + tensor expand_dims_81 = const()[name = string("expand_dims_81"), val = tensor([0])]; + tensor expand_dims_82 = const()[name = string("expand_dims_82"), val = tensor([49])]; + int32 concat_114_axis_0 = const()[name = string("concat_114_axis_0"), val = int32(0)]; + bool concat_114_interleave_0 = const()[name = string("concat_114_interleave_0"), val = bool(false)]; + tensor concat_114 = concat(axis = concat_114_axis_0, interleave = concat_114_interleave_0, values = (expand_dims_78, expand_dims_79, current_pos, expand_dims_81))[name = string("concat_114")]; + tensor concat_115_values1_0 = const()[name = string("concat_115_values1_0"), val = tensor([0])]; + tensor concat_115_values3_0 = const()[name = string("concat_115_values3_0"), val = tensor([0])]; + int32 concat_115_axis_0 = const()[name = string("concat_115_axis_0"), val = int32(0)]; + bool concat_115_interleave_0 = const()[name = string("concat_115_interleave_0"), val = bool(false)]; + tensor concat_115 = concat(axis = concat_115_axis_0, interleave = concat_115_interleave_0, values = (expand_dims_82, concat_115_values1_0, var_1042, concat_115_values3_0))[name = string("concat_115")]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_51 = transpose(perm = var_4143, x = var_4138)[name = string("transpose_68")]; + tensor model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_114, begin_mask = model_model_kv_cache_0_internal_tensor_assign_14_begin_mask_0, end = concat_115, end_mask = model_model_kv_cache_0_internal_tensor_assign_14_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_14_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_14_stride_0, update = value_states_51, x = coreml_update_state_40)[name = string("model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_14_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_97_write_state")]; + tensor coreml_update_state_41 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_97")]; + tensor var_4331_begin_0 = const()[name = string("op_4331_begin_0"), val = tensor([20, 0, 0, 0])]; + tensor var_4331_end_0 = const()[name = string("op_4331_end_0"), val = tensor([21, 8, 1024, 128])]; + tensor var_4331_end_mask_0 = const()[name = string("op_4331_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4331_cast_fp16 = slice_by_index(begin = var_4331_begin_0, end = var_4331_end_0, end_mask = var_4331_end_mask_0, x = coreml_update_state_41)[name = string("op_4331_cast_fp16")]; + tensor K_layer_cache_13_axes_0 = const()[name = string("K_layer_cache_13_axes_0"), val = tensor([0])]; + tensor K_layer_cache_13_cast_fp16 = squeeze(axes = K_layer_cache_13_axes_0, x = var_4331_cast_fp16)[name = string("K_layer_cache_13_cast_fp16")]; + tensor var_4338_begin_0 = const()[name = string("op_4338_begin_0"), val = tensor([48, 0, 0, 0])]; + tensor var_4338_end_0 = const()[name = string("op_4338_end_0"), val = tensor([49, 8, 1024, 128])]; + tensor var_4338_end_mask_0 = const()[name = string("op_4338_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4338_cast_fp16 = slice_by_index(begin = var_4338_begin_0, end = var_4338_end_0, end_mask = var_4338_end_mask_0, x = coreml_update_state_41)[name = string("op_4338_cast_fp16")]; + tensor V_layer_cache_13_axes_0 = const()[name = string("V_layer_cache_13_axes_0"), val = tensor([0])]; + tensor V_layer_cache_13_cast_fp16 = squeeze(axes = V_layer_cache_13_axes_0, x = var_4338_cast_fp16)[name = string("V_layer_cache_13_cast_fp16")]; + tensor x_99_axes_0 = const()[name = string("x_99_axes_0"), val = tensor([1])]; + tensor x_99_cast_fp16 = expand_dims(axes = x_99_axes_0, x = K_layer_cache_13_cast_fp16)[name = string("x_99_cast_fp16")]; + tensor var_4367 = const()[name = string("op_4367"), val = tensor([1, 2, 1, 1])]; + tensor x_101_cast_fp16 = tile(reps = var_4367, x = x_99_cast_fp16)[name = string("x_101_cast_fp16")]; + tensor var_4379 = const()[name = string("op_4379"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_67_cast_fp16 = reshape(shape = var_4379, x = x_101_cast_fp16)[name = string("key_states_67_cast_fp16")]; + tensor x_105_axes_0 = const()[name = string("x_105_axes_0"), val = tensor([1])]; + tensor x_105_cast_fp16 = expand_dims(axes = x_105_axes_0, x = V_layer_cache_13_cast_fp16)[name = string("x_105_cast_fp16")]; + tensor var_4387 = const()[name = string("op_4387"), val = tensor([1, 2, 1, 1])]; + tensor x_107_cast_fp16 = tile(reps = var_4387, x = x_105_cast_fp16)[name = string("x_107_cast_fp16")]; + bool var_4414_transpose_x_0 = const()[name = string("op_4414_transpose_x_0"), val = bool(false)]; + bool var_4414_transpose_y_0 = const()[name = string("op_4414_transpose_y_0"), val = bool(true)]; + tensor var_4414 = matmul(transpose_x = var_4414_transpose_x_0, transpose_y = var_4414_transpose_y_0, x = query_states_51, y = key_states_67_cast_fp16)[name = string("op_4414")]; + fp16 var_4415_to_fp16 = const()[name = string("op_4415_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_25_cast_fp16 = mul(x = var_4414, y = var_4415_to_fp16)[name = string("attn_weights_25_cast_fp16")]; + tensor attn_weights_27_cast_fp16 = add(x = attn_weights_25_cast_fp16, y = causal_mask)[name = string("attn_weights_27_cast_fp16")]; + int32 var_4450 = const()[name = string("op_4450"), val = int32(-1)]; + tensor var_4452_cast_fp16 = softmax(axis = var_4450, x = attn_weights_27_cast_fp16)[name = string("op_4452_cast_fp16")]; + tensor concat_120 = const()[name = string("concat_120"), val = tensor([16, 128, 1024])]; + tensor reshape_18_cast_fp16 = reshape(shape = concat_120, x = var_4452_cast_fp16)[name = string("reshape_18_cast_fp16")]; + tensor concat_121 = const()[name = string("concat_121"), val = tensor([16, 1024, 128])]; + tensor reshape_19_cast_fp16 = reshape(shape = concat_121, x = x_107_cast_fp16)[name = string("reshape_19_cast_fp16")]; + bool matmul_6_transpose_x_0 = const()[name = string("matmul_6_transpose_x_0"), val = bool(false)]; + bool matmul_6_transpose_y_0 = const()[name = string("matmul_6_transpose_y_0"), val = bool(false)]; + tensor matmul_6_cast_fp16 = matmul(transpose_x = matmul_6_transpose_x_0, transpose_y = matmul_6_transpose_y_0, x = reshape_18_cast_fp16, y = reshape_19_cast_fp16)[name = string("matmul_6_cast_fp16")]; + tensor concat_125 = const()[name = string("concat_125"), val = tensor([1, 16, 128, 128])]; + tensor reshape_20_cast_fp16 = reshape(shape = concat_125, x = matmul_6_cast_fp16)[name = string("reshape_20_cast_fp16")]; + tensor var_4464_perm_0 = const()[name = string("op_4464_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_4483 = const()[name = string("op_4483"), val = tensor([1, 128, 2048])]; + tensor var_4464_cast_fp16 = transpose(perm = var_4464_perm_0, x = reshape_20_cast_fp16)[name = string("transpose_67")]; + tensor attn_output_65_cast_fp16 = reshape(shape = var_4483, x = var_4464_cast_fp16)[name = string("attn_output_65_cast_fp16")]; + tensor var_4488 = const()[name = string("op_4488"), val = tensor([0, 2, 1])]; + string var_4504_pad_type_0 = const()[name = string("op_4504_pad_type_0"), val = string("valid")]; + int32 var_4504_groups_0 = const()[name = string("op_4504_groups_0"), val = int32(1)]; + tensor var_4504_strides_0 = const()[name = string("op_4504_strides_0"), val = tensor([1])]; + tensor var_4504_pad_0 = const()[name = string("op_4504_pad_0"), val = tensor([0, 0])]; + tensor var_4504_dilations_0 = const()[name = string("op_4504_dilations_0"), val = tensor([1])]; + tensor squeeze_6_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689508992))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693703360))))[name = string("squeeze_6_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_4489_cast_fp16 = transpose(perm = var_4488, x = attn_output_65_cast_fp16)[name = string("transpose_66")]; + tensor var_4504_cast_fp16 = conv(dilations = var_4504_dilations_0, groups = var_4504_groups_0, pad = var_4504_pad_0, pad_type = var_4504_pad_type_0, strides = var_4504_strides_0, weight = squeeze_6_cast_fp16_to_fp32_to_fp16_palettized, x = var_4489_cast_fp16)[name = string("op_4504_cast_fp16")]; + tensor var_4508 = const()[name = string("op_4508"), val = tensor([0, 2, 1])]; + tensor attn_output_69_cast_fp16 = transpose(perm = var_4508, x = var_4504_cast_fp16)[name = string("transpose_65")]; + tensor hidden_states_69_cast_fp16 = add(x = hidden_states_61_cast_fp16, y = attn_output_69_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; + int32 var_4521 = const()[name = string("op_4521"), val = int32(-1)]; + fp16 const_235_promoted_to_fp16 = const()[name = string("const_235_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4523_cast_fp16 = mul(x = hidden_states_69_cast_fp16, y = const_235_promoted_to_fp16)[name = string("op_4523_cast_fp16")]; + bool input_119_interleave_0 = const()[name = string("input_119_interleave_0"), val = bool(false)]; + tensor input_119_cast_fp16 = concat(axis = var_4521, interleave = input_119_interleave_0, values = (hidden_states_69_cast_fp16, var_4523_cast_fp16))[name = string("input_119_cast_fp16")]; + tensor normed_109_axes_0 = const()[name = string("normed_109_axes_0"), val = tensor([-1])]; + fp16 var_4518_to_fp16 = const()[name = string("op_4518_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_109_cast_fp16 = layer_norm(axes = normed_109_axes_0, epsilon = var_4518_to_fp16, x = input_119_cast_fp16)[name = string("normed_109_cast_fp16")]; + tensor normed_111_begin_0 = const()[name = string("normed_111_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_111_end_0 = const()[name = string("normed_111_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_111_end_mask_0 = const()[name = string("normed_111_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_111_cast_fp16 = slice_by_index(begin = normed_111_begin_0, end = normed_111_end_0, end_mask = normed_111_end_mask_0, x = normed_109_cast_fp16)[name = string("normed_111_cast_fp16")]; + tensor const_238_promoted_to_fp16 = const()[name = string("const_238_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693834496)))]; + tensor x_109_cast_fp16 = mul(x = normed_111_cast_fp16, y = const_238_promoted_to_fp16)[name = string("x_109_cast_fp16")]; + tensor var_4548 = const()[name = string("op_4548"), val = tensor([0, 2, 1])]; + tensor input_121_axes_0 = const()[name = string("input_121_axes_0"), val = tensor([2])]; + tensor var_4549 = transpose(perm = var_4548, x = x_109_cast_fp16)[name = string("transpose_64")]; + tensor input_121 = expand_dims(axes = input_121_axes_0, x = var_4549)[name = string("input_121")]; + string input_123_pad_type_0 = const()[name = string("input_123_pad_type_0"), val = string("valid")]; + tensor input_123_strides_0 = const()[name = string("input_123_strides_0"), val = tensor([1, 1])]; + tensor input_123_pad_0 = const()[name = string("input_123_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_123_dilations_0 = const()[name = string("input_123_dilations_0"), val = tensor([1, 1])]; + int32 input_123_groups_0 = const()[name = string("input_123_groups_0"), val = int32(1)]; + tensor input_123 = conv(dilations = input_123_dilations_0, groups = input_123_groups_0, pad = input_123_pad_0, pad_type = input_123_pad_type_0, strides = input_123_strides_0, weight = model_model_layers_20_mlp_gate_proj_weight_palettized, x = input_121)[name = string("input_123")]; + string b_13_pad_type_0 = const()[name = string("b_13_pad_type_0"), val = string("valid")]; + tensor b_13_strides_0 = const()[name = string("b_13_strides_0"), val = tensor([1, 1])]; + tensor b_13_pad_0 = const()[name = string("b_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_13_dilations_0 = const()[name = string("b_13_dilations_0"), val = tensor([1, 1])]; + int32 b_13_groups_0 = const()[name = string("b_13_groups_0"), val = int32(1)]; + tensor b_13 = conv(dilations = b_13_dilations_0, groups = b_13_groups_0, pad = b_13_pad_0, pad_type = b_13_pad_type_0, strides = b_13_strides_0, weight = model_model_layers_20_mlp_up_proj_weight_palettized, x = input_121)[name = string("b_13")]; + tensor c_13 = silu(x = input_123)[name = string("c_13")]; + tensor input_125 = mul(x = c_13, y = b_13)[name = string("input_125")]; + string e_13_pad_type_0 = const()[name = string("e_13_pad_type_0"), val = string("valid")]; + tensor e_13_strides_0 = const()[name = string("e_13_strides_0"), val = tensor([1, 1])]; + tensor e_13_pad_0 = const()[name = string("e_13_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_13_dilations_0 = const()[name = string("e_13_dilations_0"), val = tensor([1, 1])]; + int32 e_13_groups_0 = const()[name = string("e_13_groups_0"), val = int32(1)]; + tensor e_13 = conv(dilations = e_13_dilations_0, groups = e_13_groups_0, pad = e_13_pad_0, pad_type = e_13_pad_type_0, strides = e_13_strides_0, weight = model_model_layers_20_mlp_down_proj_weight_palettized, x = input_125)[name = string("e_13")]; + tensor var_4571_axes_0 = const()[name = string("op_4571_axes_0"), val = tensor([2])]; + tensor var_4571 = squeeze(axes = var_4571_axes_0, x = e_13)[name = string("op_4571")]; + tensor var_4572 = const()[name = string("op_4572"), val = tensor([0, 2, 1])]; + tensor var_4573 = transpose(perm = var_4572, x = var_4571)[name = string("transpose_63")]; + tensor hidden_states_71_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = var_4573)[name = string("hidden_states_71_cast_fp16")]; + int32 var_4585 = const()[name = string("op_4585"), val = int32(-1)]; + fp16 const_239_promoted_to_fp16 = const()[name = string("const_239_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_4587_cast_fp16 = mul(x = hidden_states_71_cast_fp16, y = const_239_promoted_to_fp16)[name = string("op_4587_cast_fp16")]; + bool input_127_interleave_0 = const()[name = string("input_127_interleave_0"), val = bool(false)]; + tensor input_127_cast_fp16 = concat(axis = var_4585, interleave = input_127_interleave_0, values = (hidden_states_71_cast_fp16, var_4587_cast_fp16))[name = string("input_127_cast_fp16")]; + tensor normed_113_axes_0 = const()[name = string("normed_113_axes_0"), val = tensor([-1])]; + fp16 var_4582_to_fp16 = const()[name = string("op_4582_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_113_cast_fp16 = layer_norm(axes = normed_113_axes_0, epsilon = var_4582_to_fp16, x = input_127_cast_fp16)[name = string("normed_113_cast_fp16")]; + tensor normed_115_begin_0 = const()[name = string("normed_115_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_115_end_0 = const()[name = string("normed_115_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_115_end_mask_0 = const()[name = string("normed_115_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_115_cast_fp16 = slice_by_index(begin = normed_115_begin_0, end = normed_115_end_0, end_mask = normed_115_end_mask_0, x = normed_113_cast_fp16)[name = string("normed_115_cast_fp16")]; + tensor const_242_promoted_to_fp16 = const()[name = string("const_242_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693838656)))]; + tensor hidden_states_73_cast_fp16 = mul(x = normed_115_cast_fp16, y = const_242_promoted_to_fp16)[name = string("hidden_states_73_cast_fp16")]; + tensor var_4610 = const()[name = string("op_4610"), val = tensor([0, 2, 1])]; + tensor var_4613_axes_0 = const()[name = string("op_4613_axes_0"), val = tensor([2])]; + tensor var_4611_cast_fp16 = transpose(perm = var_4610, x = hidden_states_73_cast_fp16)[name = string("transpose_62")]; + tensor var_4613_cast_fp16 = expand_dims(axes = var_4613_axes_0, x = var_4611_cast_fp16)[name = string("op_4613_cast_fp16")]; + string query_states_57_pad_type_0 = const()[name = string("query_states_57_pad_type_0"), val = string("valid")]; + tensor query_states_57_strides_0 = const()[name = string("query_states_57_strides_0"), val = tensor([1, 1])]; + tensor query_states_57_pad_0 = const()[name = string("query_states_57_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_57_dilations_0 = const()[name = string("query_states_57_dilations_0"), val = tensor([1, 1])]; + int32 query_states_57_groups_0 = const()[name = string("query_states_57_groups_0"), val = int32(1)]; + tensor query_states_57 = conv(dilations = query_states_57_dilations_0, groups = query_states_57_groups_0, pad = query_states_57_pad_0, pad_type = query_states_57_pad_type_0, strides = query_states_57_strides_0, weight = model_model_layers_21_self_attn_q_proj_weight_palettized, x = var_4613_cast_fp16)[name = string("query_states_57")]; + string key_states_71_pad_type_0 = const()[name = string("key_states_71_pad_type_0"), val = string("valid")]; + tensor key_states_71_strides_0 = const()[name = string("key_states_71_strides_0"), val = tensor([1, 1])]; + tensor key_states_71_pad_0 = const()[name = string("key_states_71_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_71_dilations_0 = const()[name = string("key_states_71_dilations_0"), val = tensor([1, 1])]; + int32 key_states_71_groups_0 = const()[name = string("key_states_71_groups_0"), val = int32(1)]; + tensor key_states_71 = conv(dilations = key_states_71_dilations_0, groups = key_states_71_groups_0, pad = key_states_71_pad_0, pad_type = key_states_71_pad_type_0, strides = key_states_71_strides_0, weight = model_model_layers_21_self_attn_k_proj_weight_palettized, x = var_4613_cast_fp16)[name = string("key_states_71")]; + string value_states_57_pad_type_0 = const()[name = string("value_states_57_pad_type_0"), val = string("valid")]; + tensor value_states_57_strides_0 = const()[name = string("value_states_57_strides_0"), val = tensor([1, 1])]; + tensor value_states_57_pad_0 = const()[name = string("value_states_57_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_57_dilations_0 = const()[name = string("value_states_57_dilations_0"), val = tensor([1, 1])]; + int32 value_states_57_groups_0 = const()[name = string("value_states_57_groups_0"), val = int32(1)]; + tensor value_states_57 = conv(dilations = value_states_57_dilations_0, groups = value_states_57_groups_0, pad = value_states_57_pad_0, pad_type = value_states_57_pad_type_0, strides = value_states_57_strides_0, weight = model_model_layers_21_self_attn_v_proj_weight_palettized, x = var_4613_cast_fp16)[name = string("value_states_57")]; + tensor var_4655 = const()[name = string("op_4655"), val = tensor([1, 16, 128, 128])]; + tensor var_4656 = reshape(shape = var_4655, x = query_states_57)[name = string("op_4656")]; + tensor var_4661 = const()[name = string("op_4661"), val = tensor([0, 1, 3, 2])]; + tensor var_4666 = const()[name = string("op_4666"), val = tensor([1, 8, 128, 128])]; + tensor var_4667 = reshape(shape = var_4666, x = key_states_71)[name = string("op_4667")]; + tensor var_4672 = const()[name = string("op_4672"), val = tensor([0, 1, 3, 2])]; + tensor var_4677 = const()[name = string("op_4677"), val = tensor([1, 8, 128, 128])]; + tensor var_4678 = reshape(shape = var_4677, x = value_states_57)[name = string("op_4678")]; + tensor var_4683 = const()[name = string("op_4683"), val = tensor([0, 1, 3, 2])]; + int32 var_4694 = const()[name = string("op_4694"), val = int32(-1)]; + fp16 const_244_promoted = const()[name = string("const_244_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_75 = transpose(perm = var_4661, x = var_4656)[name = string("transpose_61")]; + tensor var_4696 = mul(x = hidden_states_75, y = const_244_promoted)[name = string("op_4696")]; + bool input_131_interleave_0 = const()[name = string("input_131_interleave_0"), val = bool(false)]; + tensor input_131 = concat(axis = var_4694, interleave = input_131_interleave_0, values = (hidden_states_75, var_4696))[name = string("input_131")]; + tensor normed_117_axes_0 = const()[name = string("normed_117_axes_0"), val = tensor([-1])]; + fp16 var_4691_to_fp16 = const()[name = string("op_4691_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_117_cast_fp16 = layer_norm(axes = normed_117_axes_0, epsilon = var_4691_to_fp16, x = input_131)[name = string("normed_117_cast_fp16")]; + tensor normed_119_begin_0 = const()[name = string("normed_119_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_119_end_0 = const()[name = string("normed_119_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_119_end_mask_0 = const()[name = string("normed_119_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_119 = slice_by_index(begin = normed_119_begin_0, end = normed_119_end_0, end_mask = normed_119_end_mask_0, x = normed_117_cast_fp16)[name = string("normed_119")]; + tensor const_247 = const()[name = string("const_247"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693842816)))]; + tensor q_15 = mul(x = normed_119, y = const_247)[name = string("q_15")]; + int32 var_4719 = const()[name = string("op_4719"), val = int32(-1)]; + fp16 const_248_promoted = const()[name = string("const_248_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_77 = transpose(perm = var_4672, x = var_4667)[name = string("transpose_60")]; + tensor var_4721 = mul(x = hidden_states_77, y = const_248_promoted)[name = string("op_4721")]; + bool input_133_interleave_0 = const()[name = string("input_133_interleave_0"), val = bool(false)]; + tensor input_133 = concat(axis = var_4719, interleave = input_133_interleave_0, values = (hidden_states_77, var_4721))[name = string("input_133")]; + tensor normed_121_axes_0 = const()[name = string("normed_121_axes_0"), val = tensor([-1])]; + fp16 var_4716_to_fp16 = const()[name = string("op_4716_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_121_cast_fp16 = layer_norm(axes = normed_121_axes_0, epsilon = var_4716_to_fp16, x = input_133)[name = string("normed_121_cast_fp16")]; + tensor normed_123_begin_0 = const()[name = string("normed_123_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_123_end_0 = const()[name = string("normed_123_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_123_end_mask_0 = const()[name = string("normed_123_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_123 = slice_by_index(begin = normed_123_begin_0, end = normed_123_end_0, end_mask = normed_123_end_mask_0, x = normed_121_cast_fp16)[name = string("normed_123")]; + tensor const_251 = const()[name = string("const_251"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693843136)))]; + tensor k_15 = mul(x = normed_123, y = const_251)[name = string("k_15")]; + tensor var_4747 = mul(x = q_15, y = cos_5)[name = string("op_4747")]; + tensor x1_29_begin_0 = const()[name = string("x1_29_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_29_end_0 = const()[name = string("x1_29_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_29_end_mask_0 = const()[name = string("x1_29_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_29 = slice_by_index(begin = x1_29_begin_0, end = x1_29_end_0, end_mask = x1_29_end_mask_0, x = q_15)[name = string("x1_29")]; + tensor x2_29_begin_0 = const()[name = string("x2_29_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_29_end_0 = const()[name = string("x2_29_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_29_end_mask_0 = const()[name = string("x2_29_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_29 = slice_by_index(begin = x2_29_begin_0, end = x2_29_end_0, end_mask = x2_29_end_mask_0, x = q_15)[name = string("x2_29")]; + fp16 const_254_promoted = const()[name = string("const_254_promoted"), val = fp16(-0x1p+0)]; + tensor var_4768 = mul(x = x2_29, y = const_254_promoted)[name = string("op_4768")]; + int32 var_4770 = const()[name = string("op_4770"), val = int32(-1)]; + bool var_4771_interleave_0 = const()[name = string("op_4771_interleave_0"), val = bool(false)]; + tensor var_4771 = concat(axis = var_4770, interleave = var_4771_interleave_0, values = (var_4768, x1_29))[name = string("op_4771")]; + tensor var_4772 = mul(x = var_4771, y = sin_5)[name = string("op_4772")]; + tensor query_states_59 = add(x = var_4747, y = var_4772)[name = string("query_states_59")]; + tensor var_4775 = mul(x = k_15, y = cos_5)[name = string("op_4775")]; + tensor x1_31_begin_0 = const()[name = string("x1_31_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_31_end_0 = const()[name = string("x1_31_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_31_end_mask_0 = const()[name = string("x1_31_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_31 = slice_by_index(begin = x1_31_begin_0, end = x1_31_end_0, end_mask = x1_31_end_mask_0, x = k_15)[name = string("x1_31")]; + tensor x2_31_begin_0 = const()[name = string("x2_31_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_31_end_0 = const()[name = string("x2_31_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_31_end_mask_0 = const()[name = string("x2_31_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_31 = slice_by_index(begin = x2_31_begin_0, end = x2_31_end_0, end_mask = x2_31_end_mask_0, x = k_15)[name = string("x2_31")]; + fp16 const_257_promoted = const()[name = string("const_257_promoted"), val = fp16(-0x1p+0)]; + tensor var_4796 = mul(x = x2_31, y = const_257_promoted)[name = string("op_4796")]; + int32 var_4798 = const()[name = string("op_4798"), val = int32(-1)]; + bool var_4799_interleave_0 = const()[name = string("op_4799_interleave_0"), val = bool(false)]; + tensor var_4799 = concat(axis = var_4798, interleave = var_4799_interleave_0, values = (var_4796, x1_31))[name = string("op_4799")]; + tensor var_4800 = mul(x = var_4799, y = sin_5)[name = string("op_4800")]; + tensor key_states_73 = add(x = var_4775, y = var_4800)[name = string("key_states_73")]; + tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([21])]; + tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; + tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; + tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([22])]; + int32 concat_128_axis_0 = const()[name = string("concat_128_axis_0"), val = int32(0)]; + bool concat_128_interleave_0 = const()[name = string("concat_128_interleave_0"), val = bool(false)]; + tensor concat_128 = concat(axis = concat_128_axis_0, interleave = concat_128_interleave_0, values = (expand_dims_84, expand_dims_85, current_pos, expand_dims_87))[name = string("concat_128")]; + tensor concat_129_values1_0 = const()[name = string("concat_129_values1_0"), val = tensor([0])]; + tensor concat_129_values3_0 = const()[name = string("concat_129_values3_0"), val = tensor([0])]; + int32 concat_129_axis_0 = const()[name = string("concat_129_axis_0"), val = int32(0)]; + bool concat_129_interleave_0 = const()[name = string("concat_129_interleave_0"), val = bool(false)]; + tensor concat_129 = concat(axis = concat_129_axis_0, interleave = concat_129_interleave_0, values = (expand_dims_88, concat_129_values1_0, var_1042, concat_129_values3_0))[name = string("concat_129")]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_128, begin_mask = model_model_kv_cache_0_internal_tensor_assign_15_begin_mask_0, end = concat_129, end_mask = model_model_kv_cache_0_internal_tensor_assign_15_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_15_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_15_stride_0, update = key_states_73, x = coreml_update_state_41)[name = string("model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_15_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_98_write_state")]; + tensor coreml_update_state_42 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_98")]; + tensor expand_dims_90 = const()[name = string("expand_dims_90"), val = tensor([49])]; + tensor expand_dims_91 = const()[name = string("expand_dims_91"), val = tensor([0])]; + tensor expand_dims_93 = const()[name = string("expand_dims_93"), val = tensor([0])]; + tensor expand_dims_94 = const()[name = string("expand_dims_94"), val = tensor([50])]; + int32 concat_132_axis_0 = const()[name = string("concat_132_axis_0"), val = int32(0)]; + bool concat_132_interleave_0 = const()[name = string("concat_132_interleave_0"), val = bool(false)]; + tensor concat_132 = concat(axis = concat_132_axis_0, interleave = concat_132_interleave_0, values = (expand_dims_90, expand_dims_91, current_pos, expand_dims_93))[name = string("concat_132")]; + tensor concat_133_values1_0 = const()[name = string("concat_133_values1_0"), val = tensor([0])]; + tensor concat_133_values3_0 = const()[name = string("concat_133_values3_0"), val = tensor([0])]; + int32 concat_133_axis_0 = const()[name = string("concat_133_axis_0"), val = int32(0)]; + bool concat_133_interleave_0 = const()[name = string("concat_133_interleave_0"), val = bool(false)]; + tensor concat_133 = concat(axis = concat_133_axis_0, interleave = concat_133_interleave_0, values = (expand_dims_94, concat_133_values1_0, var_1042, concat_133_values3_0))[name = string("concat_133")]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_59 = transpose(perm = var_4683, x = var_4678)[name = string("transpose_59")]; + tensor model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_132, begin_mask = model_model_kv_cache_0_internal_tensor_assign_16_begin_mask_0, end = concat_133, end_mask = model_model_kv_cache_0_internal_tensor_assign_16_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_16_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_16_stride_0, update = value_states_59, x = coreml_update_state_42)[name = string("model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_16_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_99_write_state")]; + tensor coreml_update_state_43 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_99")]; + tensor var_4871_begin_0 = const()[name = string("op_4871_begin_0"), val = tensor([21, 0, 0, 0])]; + tensor var_4871_end_0 = const()[name = string("op_4871_end_0"), val = tensor([22, 8, 1024, 128])]; + tensor var_4871_end_mask_0 = const()[name = string("op_4871_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4871_cast_fp16 = slice_by_index(begin = var_4871_begin_0, end = var_4871_end_0, end_mask = var_4871_end_mask_0, x = coreml_update_state_43)[name = string("op_4871_cast_fp16")]; + tensor K_layer_cache_15_axes_0 = const()[name = string("K_layer_cache_15_axes_0"), val = tensor([0])]; + tensor K_layer_cache_15_cast_fp16 = squeeze(axes = K_layer_cache_15_axes_0, x = var_4871_cast_fp16)[name = string("K_layer_cache_15_cast_fp16")]; + tensor var_4878_begin_0 = const()[name = string("op_4878_begin_0"), val = tensor([49, 0, 0, 0])]; + tensor var_4878_end_0 = const()[name = string("op_4878_end_0"), val = tensor([50, 8, 1024, 128])]; + tensor var_4878_end_mask_0 = const()[name = string("op_4878_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_4878_cast_fp16 = slice_by_index(begin = var_4878_begin_0, end = var_4878_end_0, end_mask = var_4878_end_mask_0, x = coreml_update_state_43)[name = string("op_4878_cast_fp16")]; + tensor V_layer_cache_15_axes_0 = const()[name = string("V_layer_cache_15_axes_0"), val = tensor([0])]; + tensor V_layer_cache_15_cast_fp16 = squeeze(axes = V_layer_cache_15_axes_0, x = var_4878_cast_fp16)[name = string("V_layer_cache_15_cast_fp16")]; + tensor x_115_axes_0 = const()[name = string("x_115_axes_0"), val = tensor([1])]; + tensor x_115_cast_fp16 = expand_dims(axes = x_115_axes_0, x = K_layer_cache_15_cast_fp16)[name = string("x_115_cast_fp16")]; + tensor var_4907 = const()[name = string("op_4907"), val = tensor([1, 2, 1, 1])]; + tensor x_117_cast_fp16 = tile(reps = var_4907, x = x_115_cast_fp16)[name = string("x_117_cast_fp16")]; + tensor var_4919 = const()[name = string("op_4919"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_77_cast_fp16 = reshape(shape = var_4919, x = x_117_cast_fp16)[name = string("key_states_77_cast_fp16")]; + tensor x_121_axes_0 = const()[name = string("x_121_axes_0"), val = tensor([1])]; + tensor x_121_cast_fp16 = expand_dims(axes = x_121_axes_0, x = V_layer_cache_15_cast_fp16)[name = string("x_121_cast_fp16")]; + tensor var_4927 = const()[name = string("op_4927"), val = tensor([1, 2, 1, 1])]; + tensor x_123_cast_fp16 = tile(reps = var_4927, x = x_121_cast_fp16)[name = string("x_123_cast_fp16")]; + bool var_4954_transpose_x_0 = const()[name = string("op_4954_transpose_x_0"), val = bool(false)]; + bool var_4954_transpose_y_0 = const()[name = string("op_4954_transpose_y_0"), val = bool(true)]; + tensor var_4954 = matmul(transpose_x = var_4954_transpose_x_0, transpose_y = var_4954_transpose_y_0, x = query_states_59, y = key_states_77_cast_fp16)[name = string("op_4954")]; + fp16 var_4955_to_fp16 = const()[name = string("op_4955_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_29_cast_fp16 = mul(x = var_4954, y = var_4955_to_fp16)[name = string("attn_weights_29_cast_fp16")]; + tensor attn_weights_31_cast_fp16 = add(x = attn_weights_29_cast_fp16, y = causal_mask)[name = string("attn_weights_31_cast_fp16")]; + int32 var_4990 = const()[name = string("op_4990"), val = int32(-1)]; + tensor var_4992_cast_fp16 = softmax(axis = var_4990, x = attn_weights_31_cast_fp16)[name = string("op_4992_cast_fp16")]; + tensor concat_138 = const()[name = string("concat_138"), val = tensor([16, 128, 1024])]; + tensor reshape_21_cast_fp16 = reshape(shape = concat_138, x = var_4992_cast_fp16)[name = string("reshape_21_cast_fp16")]; + tensor concat_139 = const()[name = string("concat_139"), val = tensor([16, 1024, 128])]; + tensor reshape_22_cast_fp16 = reshape(shape = concat_139, x = x_123_cast_fp16)[name = string("reshape_22_cast_fp16")]; + bool matmul_7_transpose_x_0 = const()[name = string("matmul_7_transpose_x_0"), val = bool(false)]; + bool matmul_7_transpose_y_0 = const()[name = string("matmul_7_transpose_y_0"), val = bool(false)]; + tensor matmul_7_cast_fp16 = matmul(transpose_x = matmul_7_transpose_x_0, transpose_y = matmul_7_transpose_y_0, x = reshape_21_cast_fp16, y = reshape_22_cast_fp16)[name = string("matmul_7_cast_fp16")]; + tensor concat_143 = const()[name = string("concat_143"), val = tensor([1, 16, 128, 128])]; + tensor reshape_23_cast_fp16 = reshape(shape = concat_143, x = matmul_7_cast_fp16)[name = string("reshape_23_cast_fp16")]; + tensor var_5004_perm_0 = const()[name = string("op_5004_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_5023 = const()[name = string("op_5023"), val = tensor([1, 128, 2048])]; + tensor var_5004_cast_fp16 = transpose(perm = var_5004_perm_0, x = reshape_23_cast_fp16)[name = string("transpose_58")]; + tensor attn_output_75_cast_fp16 = reshape(shape = var_5023, x = var_5004_cast_fp16)[name = string("attn_output_75_cast_fp16")]; + tensor var_5028 = const()[name = string("op_5028"), val = tensor([0, 2, 1])]; + string var_5044_pad_type_0 = const()[name = string("op_5044_pad_type_0"), val = string("valid")]; + int32 var_5044_groups_0 = const()[name = string("op_5044_groups_0"), val = int32(1)]; + tensor var_5044_strides_0 = const()[name = string("op_5044_strides_0"), val = tensor([1])]; + tensor var_5044_pad_0 = const()[name = string("op_5044_pad_0"), val = tensor([0, 0])]; + tensor var_5044_dilations_0 = const()[name = string("op_5044_dilations_0"), val = tensor([1])]; + tensor squeeze_7_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(693843456))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698037824))))[name = string("squeeze_7_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_5029_cast_fp16 = transpose(perm = var_5028, x = attn_output_75_cast_fp16)[name = string("transpose_57")]; + tensor var_5044_cast_fp16 = conv(dilations = var_5044_dilations_0, groups = var_5044_groups_0, pad = var_5044_pad_0, pad_type = var_5044_pad_type_0, strides = var_5044_strides_0, weight = squeeze_7_cast_fp16_to_fp32_to_fp16_palettized, x = var_5029_cast_fp16)[name = string("op_5044_cast_fp16")]; + tensor var_5048 = const()[name = string("op_5048"), val = tensor([0, 2, 1])]; + tensor attn_output_79_cast_fp16 = transpose(perm = var_5048, x = var_5044_cast_fp16)[name = string("transpose_56")]; + tensor hidden_states_79_cast_fp16 = add(x = hidden_states_71_cast_fp16, y = attn_output_79_cast_fp16)[name = string("hidden_states_79_cast_fp16")]; + int32 var_5061 = const()[name = string("op_5061"), val = int32(-1)]; + fp16 const_269_promoted_to_fp16 = const()[name = string("const_269_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5063_cast_fp16 = mul(x = hidden_states_79_cast_fp16, y = const_269_promoted_to_fp16)[name = string("op_5063_cast_fp16")]; + bool input_137_interleave_0 = const()[name = string("input_137_interleave_0"), val = bool(false)]; + tensor input_137_cast_fp16 = concat(axis = var_5061, interleave = input_137_interleave_0, values = (hidden_states_79_cast_fp16, var_5063_cast_fp16))[name = string("input_137_cast_fp16")]; + tensor normed_125_axes_0 = const()[name = string("normed_125_axes_0"), val = tensor([-1])]; + fp16 var_5058_to_fp16 = const()[name = string("op_5058_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_125_cast_fp16 = layer_norm(axes = normed_125_axes_0, epsilon = var_5058_to_fp16, x = input_137_cast_fp16)[name = string("normed_125_cast_fp16")]; + tensor normed_127_begin_0 = const()[name = string("normed_127_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_127_end_0 = const()[name = string("normed_127_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_127_end_mask_0 = const()[name = string("normed_127_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_127_cast_fp16 = slice_by_index(begin = normed_127_begin_0, end = normed_127_end_0, end_mask = normed_127_end_mask_0, x = normed_125_cast_fp16)[name = string("normed_127_cast_fp16")]; + tensor const_272_promoted_to_fp16 = const()[name = string("const_272_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698168960)))]; + tensor x_125_cast_fp16 = mul(x = normed_127_cast_fp16, y = const_272_promoted_to_fp16)[name = string("x_125_cast_fp16")]; + tensor var_5088 = const()[name = string("op_5088"), val = tensor([0, 2, 1])]; + tensor input_139_axes_0 = const()[name = string("input_139_axes_0"), val = tensor([2])]; + tensor var_5089 = transpose(perm = var_5088, x = x_125_cast_fp16)[name = string("transpose_55")]; + tensor input_139 = expand_dims(axes = input_139_axes_0, x = var_5089)[name = string("input_139")]; + string input_141_pad_type_0 = const()[name = string("input_141_pad_type_0"), val = string("valid")]; + tensor input_141_strides_0 = const()[name = string("input_141_strides_0"), val = tensor([1, 1])]; + tensor input_141_pad_0 = const()[name = string("input_141_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_141_dilations_0 = const()[name = string("input_141_dilations_0"), val = tensor([1, 1])]; + int32 input_141_groups_0 = const()[name = string("input_141_groups_0"), val = int32(1)]; + tensor input_141 = conv(dilations = input_141_dilations_0, groups = input_141_groups_0, pad = input_141_pad_0, pad_type = input_141_pad_type_0, strides = input_141_strides_0, weight = model_model_layers_21_mlp_gate_proj_weight_palettized, x = input_139)[name = string("input_141")]; + string b_15_pad_type_0 = const()[name = string("b_15_pad_type_0"), val = string("valid")]; + tensor b_15_strides_0 = const()[name = string("b_15_strides_0"), val = tensor([1, 1])]; + tensor b_15_pad_0 = const()[name = string("b_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_15_dilations_0 = const()[name = string("b_15_dilations_0"), val = tensor([1, 1])]; + int32 b_15_groups_0 = const()[name = string("b_15_groups_0"), val = int32(1)]; + tensor b_15 = conv(dilations = b_15_dilations_0, groups = b_15_groups_0, pad = b_15_pad_0, pad_type = b_15_pad_type_0, strides = b_15_strides_0, weight = model_model_layers_21_mlp_up_proj_weight_palettized, x = input_139)[name = string("b_15")]; + tensor c_15 = silu(x = input_141)[name = string("c_15")]; + tensor input_143 = mul(x = c_15, y = b_15)[name = string("input_143")]; + string e_15_pad_type_0 = const()[name = string("e_15_pad_type_0"), val = string("valid")]; + tensor e_15_strides_0 = const()[name = string("e_15_strides_0"), val = tensor([1, 1])]; + tensor e_15_pad_0 = const()[name = string("e_15_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_15_dilations_0 = const()[name = string("e_15_dilations_0"), val = tensor([1, 1])]; + int32 e_15_groups_0 = const()[name = string("e_15_groups_0"), val = int32(1)]; + tensor e_15 = conv(dilations = e_15_dilations_0, groups = e_15_groups_0, pad = e_15_pad_0, pad_type = e_15_pad_type_0, strides = e_15_strides_0, weight = model_model_layers_21_mlp_down_proj_weight_palettized, x = input_143)[name = string("e_15")]; + tensor var_5111_axes_0 = const()[name = string("op_5111_axes_0"), val = tensor([2])]; + tensor var_5111 = squeeze(axes = var_5111_axes_0, x = e_15)[name = string("op_5111")]; + tensor var_5112 = const()[name = string("op_5112"), val = tensor([0, 2, 1])]; + tensor var_5113 = transpose(perm = var_5112, x = var_5111)[name = string("transpose_54")]; + tensor hidden_states_81_cast_fp16 = add(x = hidden_states_79_cast_fp16, y = var_5113)[name = string("hidden_states_81_cast_fp16")]; + int32 var_5125 = const()[name = string("op_5125"), val = int32(-1)]; + fp16 const_273_promoted_to_fp16 = const()[name = string("const_273_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5127_cast_fp16 = mul(x = hidden_states_81_cast_fp16, y = const_273_promoted_to_fp16)[name = string("op_5127_cast_fp16")]; + bool input_145_interleave_0 = const()[name = string("input_145_interleave_0"), val = bool(false)]; + tensor input_145_cast_fp16 = concat(axis = var_5125, interleave = input_145_interleave_0, values = (hidden_states_81_cast_fp16, var_5127_cast_fp16))[name = string("input_145_cast_fp16")]; + tensor normed_129_axes_0 = const()[name = string("normed_129_axes_0"), val = tensor([-1])]; + fp16 var_5122_to_fp16 = const()[name = string("op_5122_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_129_cast_fp16 = layer_norm(axes = normed_129_axes_0, epsilon = var_5122_to_fp16, x = input_145_cast_fp16)[name = string("normed_129_cast_fp16")]; + tensor normed_131_begin_0 = const()[name = string("normed_131_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_131_end_0 = const()[name = string("normed_131_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_131_end_mask_0 = const()[name = string("normed_131_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_131_cast_fp16 = slice_by_index(begin = normed_131_begin_0, end = normed_131_end_0, end_mask = normed_131_end_mask_0, x = normed_129_cast_fp16)[name = string("normed_131_cast_fp16")]; + tensor const_276_promoted_to_fp16 = const()[name = string("const_276_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698173120)))]; + tensor hidden_states_83_cast_fp16 = mul(x = normed_131_cast_fp16, y = const_276_promoted_to_fp16)[name = string("hidden_states_83_cast_fp16")]; + tensor var_5150 = const()[name = string("op_5150"), val = tensor([0, 2, 1])]; + tensor var_5153_axes_0 = const()[name = string("op_5153_axes_0"), val = tensor([2])]; + tensor var_5151_cast_fp16 = transpose(perm = var_5150, x = hidden_states_83_cast_fp16)[name = string("transpose_53")]; + tensor var_5153_cast_fp16 = expand_dims(axes = var_5153_axes_0, x = var_5151_cast_fp16)[name = string("op_5153_cast_fp16")]; + string query_states_65_pad_type_0 = const()[name = string("query_states_65_pad_type_0"), val = string("valid")]; + tensor query_states_65_strides_0 = const()[name = string("query_states_65_strides_0"), val = tensor([1, 1])]; + tensor query_states_65_pad_0 = const()[name = string("query_states_65_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_65_dilations_0 = const()[name = string("query_states_65_dilations_0"), val = tensor([1, 1])]; + int32 query_states_65_groups_0 = const()[name = string("query_states_65_groups_0"), val = int32(1)]; + tensor query_states_65 = conv(dilations = query_states_65_dilations_0, groups = query_states_65_groups_0, pad = query_states_65_pad_0, pad_type = query_states_65_pad_type_0, strides = query_states_65_strides_0, weight = model_model_layers_22_self_attn_q_proj_weight_palettized, x = var_5153_cast_fp16)[name = string("query_states_65")]; + string key_states_81_pad_type_0 = const()[name = string("key_states_81_pad_type_0"), val = string("valid")]; + tensor key_states_81_strides_0 = const()[name = string("key_states_81_strides_0"), val = tensor([1, 1])]; + tensor key_states_81_pad_0 = const()[name = string("key_states_81_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_81_dilations_0 = const()[name = string("key_states_81_dilations_0"), val = tensor([1, 1])]; + int32 key_states_81_groups_0 = const()[name = string("key_states_81_groups_0"), val = int32(1)]; + tensor key_states_81 = conv(dilations = key_states_81_dilations_0, groups = key_states_81_groups_0, pad = key_states_81_pad_0, pad_type = key_states_81_pad_type_0, strides = key_states_81_strides_0, weight = model_model_layers_22_self_attn_k_proj_weight_palettized, x = var_5153_cast_fp16)[name = string("key_states_81")]; + string value_states_65_pad_type_0 = const()[name = string("value_states_65_pad_type_0"), val = string("valid")]; + tensor value_states_65_strides_0 = const()[name = string("value_states_65_strides_0"), val = tensor([1, 1])]; + tensor value_states_65_pad_0 = const()[name = string("value_states_65_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_65_dilations_0 = const()[name = string("value_states_65_dilations_0"), val = tensor([1, 1])]; + int32 value_states_65_groups_0 = const()[name = string("value_states_65_groups_0"), val = int32(1)]; + tensor value_states_65 = conv(dilations = value_states_65_dilations_0, groups = value_states_65_groups_0, pad = value_states_65_pad_0, pad_type = value_states_65_pad_type_0, strides = value_states_65_strides_0, weight = model_model_layers_22_self_attn_v_proj_weight_palettized, x = var_5153_cast_fp16)[name = string("value_states_65")]; + tensor var_5195 = const()[name = string("op_5195"), val = tensor([1, 16, 128, 128])]; + tensor var_5196 = reshape(shape = var_5195, x = query_states_65)[name = string("op_5196")]; + tensor var_5201 = const()[name = string("op_5201"), val = tensor([0, 1, 3, 2])]; + tensor var_5206 = const()[name = string("op_5206"), val = tensor([1, 8, 128, 128])]; + tensor var_5207 = reshape(shape = var_5206, x = key_states_81)[name = string("op_5207")]; + tensor var_5212 = const()[name = string("op_5212"), val = tensor([0, 1, 3, 2])]; + tensor var_5217 = const()[name = string("op_5217"), val = tensor([1, 8, 128, 128])]; + tensor var_5218 = reshape(shape = var_5217, x = value_states_65)[name = string("op_5218")]; + tensor var_5223 = const()[name = string("op_5223"), val = tensor([0, 1, 3, 2])]; + int32 var_5234 = const()[name = string("op_5234"), val = int32(-1)]; + fp16 const_278_promoted = const()[name = string("const_278_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_85 = transpose(perm = var_5201, x = var_5196)[name = string("transpose_52")]; + tensor var_5236 = mul(x = hidden_states_85, y = const_278_promoted)[name = string("op_5236")]; + bool input_149_interleave_0 = const()[name = string("input_149_interleave_0"), val = bool(false)]; + tensor input_149 = concat(axis = var_5234, interleave = input_149_interleave_0, values = (hidden_states_85, var_5236))[name = string("input_149")]; + tensor normed_133_axes_0 = const()[name = string("normed_133_axes_0"), val = tensor([-1])]; + fp16 var_5231_to_fp16 = const()[name = string("op_5231_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_133_cast_fp16 = layer_norm(axes = normed_133_axes_0, epsilon = var_5231_to_fp16, x = input_149)[name = string("normed_133_cast_fp16")]; + tensor normed_135_begin_0 = const()[name = string("normed_135_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_135_end_0 = const()[name = string("normed_135_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_135_end_mask_0 = const()[name = string("normed_135_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_135 = slice_by_index(begin = normed_135_begin_0, end = normed_135_end_0, end_mask = normed_135_end_mask_0, x = normed_133_cast_fp16)[name = string("normed_135")]; + tensor const_281 = const()[name = string("const_281"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698177280)))]; + tensor q_17 = mul(x = normed_135, y = const_281)[name = string("q_17")]; + int32 var_5259 = const()[name = string("op_5259"), val = int32(-1)]; + fp16 const_282_promoted = const()[name = string("const_282_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_87 = transpose(perm = var_5212, x = var_5207)[name = string("transpose_51")]; + tensor var_5261 = mul(x = hidden_states_87, y = const_282_promoted)[name = string("op_5261")]; + bool input_151_interleave_0 = const()[name = string("input_151_interleave_0"), val = bool(false)]; + tensor input_151 = concat(axis = var_5259, interleave = input_151_interleave_0, values = (hidden_states_87, var_5261))[name = string("input_151")]; + tensor normed_137_axes_0 = const()[name = string("normed_137_axes_0"), val = tensor([-1])]; + fp16 var_5256_to_fp16 = const()[name = string("op_5256_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_137_cast_fp16 = layer_norm(axes = normed_137_axes_0, epsilon = var_5256_to_fp16, x = input_151)[name = string("normed_137_cast_fp16")]; + tensor normed_139_begin_0 = const()[name = string("normed_139_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_139_end_0 = const()[name = string("normed_139_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_139_end_mask_0 = const()[name = string("normed_139_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_139 = slice_by_index(begin = normed_139_begin_0, end = normed_139_end_0, end_mask = normed_139_end_mask_0, x = normed_137_cast_fp16)[name = string("normed_139")]; + tensor const_285 = const()[name = string("const_285"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698177600)))]; + tensor k_17 = mul(x = normed_139, y = const_285)[name = string("k_17")]; + tensor var_5287 = mul(x = q_17, y = cos_5)[name = string("op_5287")]; + tensor x1_33_begin_0 = const()[name = string("x1_33_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_33_end_0 = const()[name = string("x1_33_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_33_end_mask_0 = const()[name = string("x1_33_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_33 = slice_by_index(begin = x1_33_begin_0, end = x1_33_end_0, end_mask = x1_33_end_mask_0, x = q_17)[name = string("x1_33")]; + tensor x2_33_begin_0 = const()[name = string("x2_33_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_33_end_0 = const()[name = string("x2_33_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_33_end_mask_0 = const()[name = string("x2_33_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_33 = slice_by_index(begin = x2_33_begin_0, end = x2_33_end_0, end_mask = x2_33_end_mask_0, x = q_17)[name = string("x2_33")]; + fp16 const_288_promoted = const()[name = string("const_288_promoted"), val = fp16(-0x1p+0)]; + tensor var_5308 = mul(x = x2_33, y = const_288_promoted)[name = string("op_5308")]; + int32 var_5310 = const()[name = string("op_5310"), val = int32(-1)]; + bool var_5311_interleave_0 = const()[name = string("op_5311_interleave_0"), val = bool(false)]; + tensor var_5311 = concat(axis = var_5310, interleave = var_5311_interleave_0, values = (var_5308, x1_33))[name = string("op_5311")]; + tensor var_5312 = mul(x = var_5311, y = sin_5)[name = string("op_5312")]; + tensor query_states_67 = add(x = var_5287, y = var_5312)[name = string("query_states_67")]; + tensor var_5315 = mul(x = k_17, y = cos_5)[name = string("op_5315")]; + tensor x1_35_begin_0 = const()[name = string("x1_35_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_35_end_0 = const()[name = string("x1_35_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_35_end_mask_0 = const()[name = string("x1_35_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_35 = slice_by_index(begin = x1_35_begin_0, end = x1_35_end_0, end_mask = x1_35_end_mask_0, x = k_17)[name = string("x1_35")]; + tensor x2_35_begin_0 = const()[name = string("x2_35_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_35_end_0 = const()[name = string("x2_35_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_35_end_mask_0 = const()[name = string("x2_35_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_35 = slice_by_index(begin = x2_35_begin_0, end = x2_35_end_0, end_mask = x2_35_end_mask_0, x = k_17)[name = string("x2_35")]; + fp16 const_291_promoted = const()[name = string("const_291_promoted"), val = fp16(-0x1p+0)]; + tensor var_5336 = mul(x = x2_35, y = const_291_promoted)[name = string("op_5336")]; + int32 var_5338 = const()[name = string("op_5338"), val = int32(-1)]; + bool var_5339_interleave_0 = const()[name = string("op_5339_interleave_0"), val = bool(false)]; + tensor var_5339 = concat(axis = var_5338, interleave = var_5339_interleave_0, values = (var_5336, x1_35))[name = string("op_5339")]; + tensor var_5340 = mul(x = var_5339, y = sin_5)[name = string("op_5340")]; + tensor key_states_83 = add(x = var_5315, y = var_5340)[name = string("key_states_83")]; + tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([22])]; + tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; + tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; + tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([23])]; + int32 concat_146_axis_0 = const()[name = string("concat_146_axis_0"), val = int32(0)]; + bool concat_146_interleave_0 = const()[name = string("concat_146_interleave_0"), val = bool(false)]; + tensor concat_146 = concat(axis = concat_146_axis_0, interleave = concat_146_interleave_0, values = (expand_dims_96, expand_dims_97, current_pos, expand_dims_99))[name = string("concat_146")]; + tensor concat_147_values1_0 = const()[name = string("concat_147_values1_0"), val = tensor([0])]; + tensor concat_147_values3_0 = const()[name = string("concat_147_values3_0"), val = tensor([0])]; + int32 concat_147_axis_0 = const()[name = string("concat_147_axis_0"), val = int32(0)]; + bool concat_147_interleave_0 = const()[name = string("concat_147_interleave_0"), val = bool(false)]; + tensor concat_147 = concat(axis = concat_147_axis_0, interleave = concat_147_interleave_0, values = (expand_dims_100, concat_147_values1_0, var_1042, concat_147_values3_0))[name = string("concat_147")]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_146, begin_mask = model_model_kv_cache_0_internal_tensor_assign_17_begin_mask_0, end = concat_147, end_mask = model_model_kv_cache_0_internal_tensor_assign_17_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_17_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_17_stride_0, update = key_states_83, x = coreml_update_state_43)[name = string("model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_17_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_100_write_state")]; + tensor coreml_update_state_44 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_100")]; + tensor expand_dims_102 = const()[name = string("expand_dims_102"), val = tensor([50])]; + tensor expand_dims_103 = const()[name = string("expand_dims_103"), val = tensor([0])]; + tensor expand_dims_105 = const()[name = string("expand_dims_105"), val = tensor([0])]; + tensor expand_dims_106 = const()[name = string("expand_dims_106"), val = tensor([51])]; + int32 concat_150_axis_0 = const()[name = string("concat_150_axis_0"), val = int32(0)]; + bool concat_150_interleave_0 = const()[name = string("concat_150_interleave_0"), val = bool(false)]; + tensor concat_150 = concat(axis = concat_150_axis_0, interleave = concat_150_interleave_0, values = (expand_dims_102, expand_dims_103, current_pos, expand_dims_105))[name = string("concat_150")]; + tensor concat_151_values1_0 = const()[name = string("concat_151_values1_0"), val = tensor([0])]; + tensor concat_151_values3_0 = const()[name = string("concat_151_values3_0"), val = tensor([0])]; + int32 concat_151_axis_0 = const()[name = string("concat_151_axis_0"), val = int32(0)]; + bool concat_151_interleave_0 = const()[name = string("concat_151_interleave_0"), val = bool(false)]; + tensor concat_151 = concat(axis = concat_151_axis_0, interleave = concat_151_interleave_0, values = (expand_dims_106, concat_151_values1_0, var_1042, concat_151_values3_0))[name = string("concat_151")]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_67 = transpose(perm = var_5223, x = var_5218)[name = string("transpose_50")]; + tensor model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_150, begin_mask = model_model_kv_cache_0_internal_tensor_assign_18_begin_mask_0, end = concat_151, end_mask = model_model_kv_cache_0_internal_tensor_assign_18_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_18_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_18_stride_0, update = value_states_67, x = coreml_update_state_44)[name = string("model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_18_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_101_write_state")]; + tensor coreml_update_state_45 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_101")]; + tensor var_5411_begin_0 = const()[name = string("op_5411_begin_0"), val = tensor([22, 0, 0, 0])]; + tensor var_5411_end_0 = const()[name = string("op_5411_end_0"), val = tensor([23, 8, 1024, 128])]; + tensor var_5411_end_mask_0 = const()[name = string("op_5411_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_5411_cast_fp16 = slice_by_index(begin = var_5411_begin_0, end = var_5411_end_0, end_mask = var_5411_end_mask_0, x = coreml_update_state_45)[name = string("op_5411_cast_fp16")]; + tensor K_layer_cache_17_axes_0 = const()[name = string("K_layer_cache_17_axes_0"), val = tensor([0])]; + tensor K_layer_cache_17_cast_fp16 = squeeze(axes = K_layer_cache_17_axes_0, x = var_5411_cast_fp16)[name = string("K_layer_cache_17_cast_fp16")]; + tensor var_5418_begin_0 = const()[name = string("op_5418_begin_0"), val = tensor([50, 0, 0, 0])]; + tensor var_5418_end_0 = const()[name = string("op_5418_end_0"), val = tensor([51, 8, 1024, 128])]; + tensor var_5418_end_mask_0 = const()[name = string("op_5418_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_5418_cast_fp16 = slice_by_index(begin = var_5418_begin_0, end = var_5418_end_0, end_mask = var_5418_end_mask_0, x = coreml_update_state_45)[name = string("op_5418_cast_fp16")]; + tensor V_layer_cache_17_axes_0 = const()[name = string("V_layer_cache_17_axes_0"), val = tensor([0])]; + tensor V_layer_cache_17_cast_fp16 = squeeze(axes = V_layer_cache_17_axes_0, x = var_5418_cast_fp16)[name = string("V_layer_cache_17_cast_fp16")]; + tensor x_131_axes_0 = const()[name = string("x_131_axes_0"), val = tensor([1])]; + tensor x_131_cast_fp16 = expand_dims(axes = x_131_axes_0, x = K_layer_cache_17_cast_fp16)[name = string("x_131_cast_fp16")]; + tensor var_5447 = const()[name = string("op_5447"), val = tensor([1, 2, 1, 1])]; + tensor x_133_cast_fp16 = tile(reps = var_5447, x = x_131_cast_fp16)[name = string("x_133_cast_fp16")]; + tensor var_5459 = const()[name = string("op_5459"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_87_cast_fp16 = reshape(shape = var_5459, x = x_133_cast_fp16)[name = string("key_states_87_cast_fp16")]; + tensor x_137_axes_0 = const()[name = string("x_137_axes_0"), val = tensor([1])]; + tensor x_137_cast_fp16 = expand_dims(axes = x_137_axes_0, x = V_layer_cache_17_cast_fp16)[name = string("x_137_cast_fp16")]; + tensor var_5467 = const()[name = string("op_5467"), val = tensor([1, 2, 1, 1])]; + tensor x_139_cast_fp16 = tile(reps = var_5467, x = x_137_cast_fp16)[name = string("x_139_cast_fp16")]; + bool var_5494_transpose_x_0 = const()[name = string("op_5494_transpose_x_0"), val = bool(false)]; + bool var_5494_transpose_y_0 = const()[name = string("op_5494_transpose_y_0"), val = bool(true)]; + tensor var_5494 = matmul(transpose_x = var_5494_transpose_x_0, transpose_y = var_5494_transpose_y_0, x = query_states_67, y = key_states_87_cast_fp16)[name = string("op_5494")]; + fp16 var_5495_to_fp16 = const()[name = string("op_5495_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_33_cast_fp16 = mul(x = var_5494, y = var_5495_to_fp16)[name = string("attn_weights_33_cast_fp16")]; + tensor attn_weights_35_cast_fp16 = add(x = attn_weights_33_cast_fp16, y = causal_mask)[name = string("attn_weights_35_cast_fp16")]; + int32 var_5530 = const()[name = string("op_5530"), val = int32(-1)]; + tensor var_5532_cast_fp16 = softmax(axis = var_5530, x = attn_weights_35_cast_fp16)[name = string("op_5532_cast_fp16")]; + tensor concat_156 = const()[name = string("concat_156"), val = tensor([16, 128, 1024])]; + tensor reshape_24_cast_fp16 = reshape(shape = concat_156, x = var_5532_cast_fp16)[name = string("reshape_24_cast_fp16")]; + tensor concat_157 = const()[name = string("concat_157"), val = tensor([16, 1024, 128])]; + tensor reshape_25_cast_fp16 = reshape(shape = concat_157, x = x_139_cast_fp16)[name = string("reshape_25_cast_fp16")]; + bool matmul_8_transpose_x_0 = const()[name = string("matmul_8_transpose_x_0"), val = bool(false)]; + bool matmul_8_transpose_y_0 = const()[name = string("matmul_8_transpose_y_0"), val = bool(false)]; + tensor matmul_8_cast_fp16 = matmul(transpose_x = matmul_8_transpose_x_0, transpose_y = matmul_8_transpose_y_0, x = reshape_24_cast_fp16, y = reshape_25_cast_fp16)[name = string("matmul_8_cast_fp16")]; + tensor concat_161 = const()[name = string("concat_161"), val = tensor([1, 16, 128, 128])]; + tensor reshape_26_cast_fp16 = reshape(shape = concat_161, x = matmul_8_cast_fp16)[name = string("reshape_26_cast_fp16")]; + tensor var_5544_perm_0 = const()[name = string("op_5544_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_5563 = const()[name = string("op_5563"), val = tensor([1, 128, 2048])]; + tensor var_5544_cast_fp16 = transpose(perm = var_5544_perm_0, x = reshape_26_cast_fp16)[name = string("transpose_49")]; + tensor attn_output_85_cast_fp16 = reshape(shape = var_5563, x = var_5544_cast_fp16)[name = string("attn_output_85_cast_fp16")]; + tensor var_5568 = const()[name = string("op_5568"), val = tensor([0, 2, 1])]; + string var_5584_pad_type_0 = const()[name = string("op_5584_pad_type_0"), val = string("valid")]; + int32 var_5584_groups_0 = const()[name = string("op_5584_groups_0"), val = int32(1)]; + tensor var_5584_strides_0 = const()[name = string("op_5584_strides_0"), val = tensor([1])]; + tensor var_5584_pad_0 = const()[name = string("op_5584_pad_0"), val = tensor([0, 0])]; + tensor var_5584_dilations_0 = const()[name = string("op_5584_dilations_0"), val = tensor([1])]; + tensor squeeze_8_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(698177920))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702372288))))[name = string("squeeze_8_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_5569_cast_fp16 = transpose(perm = var_5568, x = attn_output_85_cast_fp16)[name = string("transpose_48")]; + tensor var_5584_cast_fp16 = conv(dilations = var_5584_dilations_0, groups = var_5584_groups_0, pad = var_5584_pad_0, pad_type = var_5584_pad_type_0, strides = var_5584_strides_0, weight = squeeze_8_cast_fp16_to_fp32_to_fp16_palettized, x = var_5569_cast_fp16)[name = string("op_5584_cast_fp16")]; + tensor var_5588 = const()[name = string("op_5588"), val = tensor([0, 2, 1])]; + tensor attn_output_89_cast_fp16 = transpose(perm = var_5588, x = var_5584_cast_fp16)[name = string("transpose_47")]; + tensor hidden_states_89_cast_fp16 = add(x = hidden_states_81_cast_fp16, y = attn_output_89_cast_fp16)[name = string("hidden_states_89_cast_fp16")]; + int32 var_5601 = const()[name = string("op_5601"), val = int32(-1)]; + fp16 const_303_promoted_to_fp16 = const()[name = string("const_303_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5603_cast_fp16 = mul(x = hidden_states_89_cast_fp16, y = const_303_promoted_to_fp16)[name = string("op_5603_cast_fp16")]; + bool input_155_interleave_0 = const()[name = string("input_155_interleave_0"), val = bool(false)]; + tensor input_155_cast_fp16 = concat(axis = var_5601, interleave = input_155_interleave_0, values = (hidden_states_89_cast_fp16, var_5603_cast_fp16))[name = string("input_155_cast_fp16")]; + tensor normed_141_axes_0 = const()[name = string("normed_141_axes_0"), val = tensor([-1])]; + fp16 var_5598_to_fp16 = const()[name = string("op_5598_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_141_cast_fp16 = layer_norm(axes = normed_141_axes_0, epsilon = var_5598_to_fp16, x = input_155_cast_fp16)[name = string("normed_141_cast_fp16")]; + tensor normed_143_begin_0 = const()[name = string("normed_143_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_143_end_0 = const()[name = string("normed_143_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_143_end_mask_0 = const()[name = string("normed_143_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_143_cast_fp16 = slice_by_index(begin = normed_143_begin_0, end = normed_143_end_0, end_mask = normed_143_end_mask_0, x = normed_141_cast_fp16)[name = string("normed_143_cast_fp16")]; + tensor const_306_promoted_to_fp16 = const()[name = string("const_306_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702503424)))]; + tensor x_141_cast_fp16 = mul(x = normed_143_cast_fp16, y = const_306_promoted_to_fp16)[name = string("x_141_cast_fp16")]; + tensor var_5628 = const()[name = string("op_5628"), val = tensor([0, 2, 1])]; + tensor input_157_axes_0 = const()[name = string("input_157_axes_0"), val = tensor([2])]; + tensor var_5629 = transpose(perm = var_5628, x = x_141_cast_fp16)[name = string("transpose_46")]; + tensor input_157 = expand_dims(axes = input_157_axes_0, x = var_5629)[name = string("input_157")]; + string input_159_pad_type_0 = const()[name = string("input_159_pad_type_0"), val = string("valid")]; + tensor input_159_strides_0 = const()[name = string("input_159_strides_0"), val = tensor([1, 1])]; + tensor input_159_pad_0 = const()[name = string("input_159_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_159_dilations_0 = const()[name = string("input_159_dilations_0"), val = tensor([1, 1])]; + int32 input_159_groups_0 = const()[name = string("input_159_groups_0"), val = int32(1)]; + tensor input_159 = conv(dilations = input_159_dilations_0, groups = input_159_groups_0, pad = input_159_pad_0, pad_type = input_159_pad_type_0, strides = input_159_strides_0, weight = model_model_layers_22_mlp_gate_proj_weight_palettized, x = input_157)[name = string("input_159")]; + string b_17_pad_type_0 = const()[name = string("b_17_pad_type_0"), val = string("valid")]; + tensor b_17_strides_0 = const()[name = string("b_17_strides_0"), val = tensor([1, 1])]; + tensor b_17_pad_0 = const()[name = string("b_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_17_dilations_0 = const()[name = string("b_17_dilations_0"), val = tensor([1, 1])]; + int32 b_17_groups_0 = const()[name = string("b_17_groups_0"), val = int32(1)]; + tensor b_17 = conv(dilations = b_17_dilations_0, groups = b_17_groups_0, pad = b_17_pad_0, pad_type = b_17_pad_type_0, strides = b_17_strides_0, weight = model_model_layers_22_mlp_up_proj_weight_palettized, x = input_157)[name = string("b_17")]; + tensor c_17 = silu(x = input_159)[name = string("c_17")]; + tensor input_161 = mul(x = c_17, y = b_17)[name = string("input_161")]; + string e_17_pad_type_0 = const()[name = string("e_17_pad_type_0"), val = string("valid")]; + tensor e_17_strides_0 = const()[name = string("e_17_strides_0"), val = tensor([1, 1])]; + tensor e_17_pad_0 = const()[name = string("e_17_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_17_dilations_0 = const()[name = string("e_17_dilations_0"), val = tensor([1, 1])]; + int32 e_17_groups_0 = const()[name = string("e_17_groups_0"), val = int32(1)]; + tensor e_17 = conv(dilations = e_17_dilations_0, groups = e_17_groups_0, pad = e_17_pad_0, pad_type = e_17_pad_type_0, strides = e_17_strides_0, weight = model_model_layers_22_mlp_down_proj_weight_palettized, x = input_161)[name = string("e_17")]; + tensor var_5651_axes_0 = const()[name = string("op_5651_axes_0"), val = tensor([2])]; + tensor var_5651 = squeeze(axes = var_5651_axes_0, x = e_17)[name = string("op_5651")]; + tensor var_5652 = const()[name = string("op_5652"), val = tensor([0, 2, 1])]; + tensor var_5653 = transpose(perm = var_5652, x = var_5651)[name = string("transpose_45")]; + tensor hidden_states_91_cast_fp16 = add(x = hidden_states_89_cast_fp16, y = var_5653)[name = string("hidden_states_91_cast_fp16")]; + int32 var_5665 = const()[name = string("op_5665"), val = int32(-1)]; + fp16 const_307_promoted_to_fp16 = const()[name = string("const_307_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_5667_cast_fp16 = mul(x = hidden_states_91_cast_fp16, y = const_307_promoted_to_fp16)[name = string("op_5667_cast_fp16")]; + bool input_163_interleave_0 = const()[name = string("input_163_interleave_0"), val = bool(false)]; + tensor input_163_cast_fp16 = concat(axis = var_5665, interleave = input_163_interleave_0, values = (hidden_states_91_cast_fp16, var_5667_cast_fp16))[name = string("input_163_cast_fp16")]; + tensor normed_145_axes_0 = const()[name = string("normed_145_axes_0"), val = tensor([-1])]; + fp16 var_5662_to_fp16 = const()[name = string("op_5662_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_145_cast_fp16 = layer_norm(axes = normed_145_axes_0, epsilon = var_5662_to_fp16, x = input_163_cast_fp16)[name = string("normed_145_cast_fp16")]; + tensor normed_147_begin_0 = const()[name = string("normed_147_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_147_end_0 = const()[name = string("normed_147_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_147_end_mask_0 = const()[name = string("normed_147_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_147_cast_fp16 = slice_by_index(begin = normed_147_begin_0, end = normed_147_end_0, end_mask = normed_147_end_mask_0, x = normed_145_cast_fp16)[name = string("normed_147_cast_fp16")]; + tensor const_310_promoted_to_fp16 = const()[name = string("const_310_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702507584)))]; + tensor hidden_states_93_cast_fp16 = mul(x = normed_147_cast_fp16, y = const_310_promoted_to_fp16)[name = string("hidden_states_93_cast_fp16")]; + tensor var_5690 = const()[name = string("op_5690"), val = tensor([0, 2, 1])]; + tensor var_5693_axes_0 = const()[name = string("op_5693_axes_0"), val = tensor([2])]; + tensor var_5691_cast_fp16 = transpose(perm = var_5690, x = hidden_states_93_cast_fp16)[name = string("transpose_44")]; + tensor var_5693_cast_fp16 = expand_dims(axes = var_5693_axes_0, x = var_5691_cast_fp16)[name = string("op_5693_cast_fp16")]; + string query_states_73_pad_type_0 = const()[name = string("query_states_73_pad_type_0"), val = string("valid")]; + tensor query_states_73_strides_0 = const()[name = string("query_states_73_strides_0"), val = tensor([1, 1])]; + tensor query_states_73_pad_0 = const()[name = string("query_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_73_dilations_0 = const()[name = string("query_states_73_dilations_0"), val = tensor([1, 1])]; + int32 query_states_73_groups_0 = const()[name = string("query_states_73_groups_0"), val = int32(1)]; + tensor query_states_73 = conv(dilations = query_states_73_dilations_0, groups = query_states_73_groups_0, pad = query_states_73_pad_0, pad_type = query_states_73_pad_type_0, strides = query_states_73_strides_0, weight = model_model_layers_23_self_attn_q_proj_weight_palettized, x = var_5693_cast_fp16)[name = string("query_states_73")]; + string key_states_91_pad_type_0 = const()[name = string("key_states_91_pad_type_0"), val = string("valid")]; + tensor key_states_91_strides_0 = const()[name = string("key_states_91_strides_0"), val = tensor([1, 1])]; + tensor key_states_91_pad_0 = const()[name = string("key_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_91_dilations_0 = const()[name = string("key_states_91_dilations_0"), val = tensor([1, 1])]; + int32 key_states_91_groups_0 = const()[name = string("key_states_91_groups_0"), val = int32(1)]; + tensor key_states_91 = conv(dilations = key_states_91_dilations_0, groups = key_states_91_groups_0, pad = key_states_91_pad_0, pad_type = key_states_91_pad_type_0, strides = key_states_91_strides_0, weight = model_model_layers_23_self_attn_k_proj_weight_palettized, x = var_5693_cast_fp16)[name = string("key_states_91")]; + string value_states_73_pad_type_0 = const()[name = string("value_states_73_pad_type_0"), val = string("valid")]; + tensor value_states_73_strides_0 = const()[name = string("value_states_73_strides_0"), val = tensor([1, 1])]; + tensor value_states_73_pad_0 = const()[name = string("value_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_73_dilations_0 = const()[name = string("value_states_73_dilations_0"), val = tensor([1, 1])]; + int32 value_states_73_groups_0 = const()[name = string("value_states_73_groups_0"), val = int32(1)]; + tensor value_states_73 = conv(dilations = value_states_73_dilations_0, groups = value_states_73_groups_0, pad = value_states_73_pad_0, pad_type = value_states_73_pad_type_0, strides = value_states_73_strides_0, weight = model_model_layers_23_self_attn_v_proj_weight_palettized, x = var_5693_cast_fp16)[name = string("value_states_73")]; + tensor var_5735 = const()[name = string("op_5735"), val = tensor([1, 16, 128, 128])]; + tensor var_5736 = reshape(shape = var_5735, x = query_states_73)[name = string("op_5736")]; + tensor var_5741 = const()[name = string("op_5741"), val = tensor([0, 1, 3, 2])]; + tensor var_5746 = const()[name = string("op_5746"), val = tensor([1, 8, 128, 128])]; + tensor var_5747 = reshape(shape = var_5746, x = key_states_91)[name = string("op_5747")]; + tensor var_5752 = const()[name = string("op_5752"), val = tensor([0, 1, 3, 2])]; + tensor var_5757 = const()[name = string("op_5757"), val = tensor([1, 8, 128, 128])]; + tensor var_5758 = reshape(shape = var_5757, x = value_states_73)[name = string("op_5758")]; + tensor var_5763 = const()[name = string("op_5763"), val = tensor([0, 1, 3, 2])]; + int32 var_5774 = const()[name = string("op_5774"), val = int32(-1)]; + fp16 const_312_promoted = const()[name = string("const_312_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_95 = transpose(perm = var_5741, x = var_5736)[name = string("transpose_43")]; + tensor var_5776 = mul(x = hidden_states_95, y = const_312_promoted)[name = string("op_5776")]; + bool input_167_interleave_0 = const()[name = string("input_167_interleave_0"), val = bool(false)]; + tensor input_167 = concat(axis = var_5774, interleave = input_167_interleave_0, values = (hidden_states_95, var_5776))[name = string("input_167")]; + tensor normed_149_axes_0 = const()[name = string("normed_149_axes_0"), val = tensor([-1])]; + fp16 var_5771_to_fp16 = const()[name = string("op_5771_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_149_cast_fp16 = layer_norm(axes = normed_149_axes_0, epsilon = var_5771_to_fp16, x = input_167)[name = string("normed_149_cast_fp16")]; + tensor normed_151_begin_0 = const()[name = string("normed_151_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_151_end_0 = const()[name = string("normed_151_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_151_end_mask_0 = const()[name = string("normed_151_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_151 = slice_by_index(begin = normed_151_begin_0, end = normed_151_end_0, end_mask = normed_151_end_mask_0, x = normed_149_cast_fp16)[name = string("normed_151")]; + tensor const_315 = const()[name = string("const_315"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702511744)))]; + tensor q_19 = mul(x = normed_151, y = const_315)[name = string("q_19")]; + int32 var_5799 = const()[name = string("op_5799"), val = int32(-1)]; + fp16 const_316_promoted = const()[name = string("const_316_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_97 = transpose(perm = var_5752, x = var_5747)[name = string("transpose_42")]; + tensor var_5801 = mul(x = hidden_states_97, y = const_316_promoted)[name = string("op_5801")]; + bool input_169_interleave_0 = const()[name = string("input_169_interleave_0"), val = bool(false)]; + tensor input_169 = concat(axis = var_5799, interleave = input_169_interleave_0, values = (hidden_states_97, var_5801))[name = string("input_169")]; + tensor normed_153_axes_0 = const()[name = string("normed_153_axes_0"), val = tensor([-1])]; + fp16 var_5796_to_fp16 = const()[name = string("op_5796_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_153_cast_fp16 = layer_norm(axes = normed_153_axes_0, epsilon = var_5796_to_fp16, x = input_169)[name = string("normed_153_cast_fp16")]; + tensor normed_155_begin_0 = const()[name = string("normed_155_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_155_end_0 = const()[name = string("normed_155_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_155_end_mask_0 = const()[name = string("normed_155_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_155 = slice_by_index(begin = normed_155_begin_0, end = normed_155_end_0, end_mask = normed_155_end_mask_0, x = normed_153_cast_fp16)[name = string("normed_155")]; + tensor const_319 = const()[name = string("const_319"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702512064)))]; + tensor k_19 = mul(x = normed_155, y = const_319)[name = string("k_19")]; + tensor var_5827 = mul(x = q_19, y = cos_5)[name = string("op_5827")]; + tensor x1_37_begin_0 = const()[name = string("x1_37_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_37_end_0 = const()[name = string("x1_37_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_37_end_mask_0 = const()[name = string("x1_37_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_37 = slice_by_index(begin = x1_37_begin_0, end = x1_37_end_0, end_mask = x1_37_end_mask_0, x = q_19)[name = string("x1_37")]; + tensor x2_37_begin_0 = const()[name = string("x2_37_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_37_end_0 = const()[name = string("x2_37_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_37_end_mask_0 = const()[name = string("x2_37_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_37 = slice_by_index(begin = x2_37_begin_0, end = x2_37_end_0, end_mask = x2_37_end_mask_0, x = q_19)[name = string("x2_37")]; + fp16 const_322_promoted = const()[name = string("const_322_promoted"), val = fp16(-0x1p+0)]; + tensor var_5848 = mul(x = x2_37, y = const_322_promoted)[name = string("op_5848")]; + int32 var_5850 = const()[name = string("op_5850"), val = int32(-1)]; + bool var_5851_interleave_0 = const()[name = string("op_5851_interleave_0"), val = bool(false)]; + tensor var_5851 = concat(axis = var_5850, interleave = var_5851_interleave_0, values = (var_5848, x1_37))[name = string("op_5851")]; + tensor var_5852 = mul(x = var_5851, y = sin_5)[name = string("op_5852")]; + tensor query_states_75 = add(x = var_5827, y = var_5852)[name = string("query_states_75")]; + tensor var_5855 = mul(x = k_19, y = cos_5)[name = string("op_5855")]; + tensor x1_39_begin_0 = const()[name = string("x1_39_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_39_end_0 = const()[name = string("x1_39_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_39_end_mask_0 = const()[name = string("x1_39_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_39 = slice_by_index(begin = x1_39_begin_0, end = x1_39_end_0, end_mask = x1_39_end_mask_0, x = k_19)[name = string("x1_39")]; + tensor x2_39_begin_0 = const()[name = string("x2_39_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_39_end_0 = const()[name = string("x2_39_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_39_end_mask_0 = const()[name = string("x2_39_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_39 = slice_by_index(begin = x2_39_begin_0, end = x2_39_end_0, end_mask = x2_39_end_mask_0, x = k_19)[name = string("x2_39")]; + fp16 const_325_promoted = const()[name = string("const_325_promoted"), val = fp16(-0x1p+0)]; + tensor var_5876 = mul(x = x2_39, y = const_325_promoted)[name = string("op_5876")]; + int32 var_5878 = const()[name = string("op_5878"), val = int32(-1)]; + bool var_5879_interleave_0 = const()[name = string("op_5879_interleave_0"), val = bool(false)]; + tensor var_5879 = concat(axis = var_5878, interleave = var_5879_interleave_0, values = (var_5876, x1_39))[name = string("op_5879")]; + tensor var_5880 = mul(x = var_5879, y = sin_5)[name = string("op_5880")]; + tensor key_states_93 = add(x = var_5855, y = var_5880)[name = string("key_states_93")]; + tensor expand_dims_108 = const()[name = string("expand_dims_108"), val = tensor([23])]; + tensor expand_dims_109 = const()[name = string("expand_dims_109"), val = tensor([0])]; + tensor expand_dims_111 = const()[name = string("expand_dims_111"), val = tensor([0])]; + tensor expand_dims_112 = const()[name = string("expand_dims_112"), val = tensor([24])]; + int32 concat_164_axis_0 = const()[name = string("concat_164_axis_0"), val = int32(0)]; + bool concat_164_interleave_0 = const()[name = string("concat_164_interleave_0"), val = bool(false)]; + tensor concat_164 = concat(axis = concat_164_axis_0, interleave = concat_164_interleave_0, values = (expand_dims_108, expand_dims_109, current_pos, expand_dims_111))[name = string("concat_164")]; + tensor concat_165_values1_0 = const()[name = string("concat_165_values1_0"), val = tensor([0])]; + tensor concat_165_values3_0 = const()[name = string("concat_165_values3_0"), val = tensor([0])]; + int32 concat_165_axis_0 = const()[name = string("concat_165_axis_0"), val = int32(0)]; + bool concat_165_interleave_0 = const()[name = string("concat_165_interleave_0"), val = bool(false)]; + tensor concat_165 = concat(axis = concat_165_axis_0, interleave = concat_165_interleave_0, values = (expand_dims_112, concat_165_values1_0, var_1042, concat_165_values3_0))[name = string("concat_165")]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_164, begin_mask = model_model_kv_cache_0_internal_tensor_assign_19_begin_mask_0, end = concat_165, end_mask = model_model_kv_cache_0_internal_tensor_assign_19_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_19_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_19_stride_0, update = key_states_93, x = coreml_update_state_45)[name = string("model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_19_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_102_write_state")]; + tensor coreml_update_state_46 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_102")]; + tensor expand_dims_114 = const()[name = string("expand_dims_114"), val = tensor([51])]; + tensor expand_dims_115 = const()[name = string("expand_dims_115"), val = tensor([0])]; + tensor expand_dims_117 = const()[name = string("expand_dims_117"), val = tensor([0])]; + tensor expand_dims_118 = const()[name = string("expand_dims_118"), val = tensor([52])]; + int32 concat_168_axis_0 = const()[name = string("concat_168_axis_0"), val = int32(0)]; + bool concat_168_interleave_0 = const()[name = string("concat_168_interleave_0"), val = bool(false)]; + tensor concat_168 = concat(axis = concat_168_axis_0, interleave = concat_168_interleave_0, values = (expand_dims_114, expand_dims_115, current_pos, expand_dims_117))[name = string("concat_168")]; + tensor concat_169_values1_0 = const()[name = string("concat_169_values1_0"), val = tensor([0])]; + tensor concat_169_values3_0 = const()[name = string("concat_169_values3_0"), val = tensor([0])]; + int32 concat_169_axis_0 = const()[name = string("concat_169_axis_0"), val = int32(0)]; + bool concat_169_interleave_0 = const()[name = string("concat_169_interleave_0"), val = bool(false)]; + tensor concat_169 = concat(axis = concat_169_axis_0, interleave = concat_169_interleave_0, values = (expand_dims_118, concat_169_values1_0, var_1042, concat_169_values3_0))[name = string("concat_169")]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_75 = transpose(perm = var_5763, x = var_5758)[name = string("transpose_41")]; + tensor model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_168, begin_mask = model_model_kv_cache_0_internal_tensor_assign_20_begin_mask_0, end = concat_169, end_mask = model_model_kv_cache_0_internal_tensor_assign_20_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_20_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_20_stride_0, update = value_states_75, x = coreml_update_state_46)[name = string("model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_20_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_103_write_state")]; + tensor coreml_update_state_47 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_103")]; + tensor var_5951_begin_0 = const()[name = string("op_5951_begin_0"), val = tensor([23, 0, 0, 0])]; + tensor var_5951_end_0 = const()[name = string("op_5951_end_0"), val = tensor([24, 8, 1024, 128])]; + tensor var_5951_end_mask_0 = const()[name = string("op_5951_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_5951_cast_fp16 = slice_by_index(begin = var_5951_begin_0, end = var_5951_end_0, end_mask = var_5951_end_mask_0, x = coreml_update_state_47)[name = string("op_5951_cast_fp16")]; + tensor K_layer_cache_19_axes_0 = const()[name = string("K_layer_cache_19_axes_0"), val = tensor([0])]; + tensor K_layer_cache_19_cast_fp16 = squeeze(axes = K_layer_cache_19_axes_0, x = var_5951_cast_fp16)[name = string("K_layer_cache_19_cast_fp16")]; + tensor var_5958_begin_0 = const()[name = string("op_5958_begin_0"), val = tensor([51, 0, 0, 0])]; + tensor var_5958_end_0 = const()[name = string("op_5958_end_0"), val = tensor([52, 8, 1024, 128])]; + tensor var_5958_end_mask_0 = const()[name = string("op_5958_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_5958_cast_fp16 = slice_by_index(begin = var_5958_begin_0, end = var_5958_end_0, end_mask = var_5958_end_mask_0, x = coreml_update_state_47)[name = string("op_5958_cast_fp16")]; + tensor V_layer_cache_19_axes_0 = const()[name = string("V_layer_cache_19_axes_0"), val = tensor([0])]; + tensor V_layer_cache_19_cast_fp16 = squeeze(axes = V_layer_cache_19_axes_0, x = var_5958_cast_fp16)[name = string("V_layer_cache_19_cast_fp16")]; + tensor x_147_axes_0 = const()[name = string("x_147_axes_0"), val = tensor([1])]; + tensor x_147_cast_fp16 = expand_dims(axes = x_147_axes_0, x = K_layer_cache_19_cast_fp16)[name = string("x_147_cast_fp16")]; + tensor var_5987 = const()[name = string("op_5987"), val = tensor([1, 2, 1, 1])]; + tensor x_149_cast_fp16 = tile(reps = var_5987, x = x_147_cast_fp16)[name = string("x_149_cast_fp16")]; + tensor var_5999 = const()[name = string("op_5999"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_97_cast_fp16 = reshape(shape = var_5999, x = x_149_cast_fp16)[name = string("key_states_97_cast_fp16")]; + tensor x_153_axes_0 = const()[name = string("x_153_axes_0"), val = tensor([1])]; + tensor x_153_cast_fp16 = expand_dims(axes = x_153_axes_0, x = V_layer_cache_19_cast_fp16)[name = string("x_153_cast_fp16")]; + tensor var_6007 = const()[name = string("op_6007"), val = tensor([1, 2, 1, 1])]; + tensor x_155_cast_fp16 = tile(reps = var_6007, x = x_153_cast_fp16)[name = string("x_155_cast_fp16")]; + bool var_6034_transpose_x_0 = const()[name = string("op_6034_transpose_x_0"), val = bool(false)]; + bool var_6034_transpose_y_0 = const()[name = string("op_6034_transpose_y_0"), val = bool(true)]; + tensor var_6034 = matmul(transpose_x = var_6034_transpose_x_0, transpose_y = var_6034_transpose_y_0, x = query_states_75, y = key_states_97_cast_fp16)[name = string("op_6034")]; + fp16 var_6035_to_fp16 = const()[name = string("op_6035_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_37_cast_fp16 = mul(x = var_6034, y = var_6035_to_fp16)[name = string("attn_weights_37_cast_fp16")]; + tensor attn_weights_39_cast_fp16 = add(x = attn_weights_37_cast_fp16, y = causal_mask)[name = string("attn_weights_39_cast_fp16")]; + int32 var_6070 = const()[name = string("op_6070"), val = int32(-1)]; + tensor var_6072_cast_fp16 = softmax(axis = var_6070, x = attn_weights_39_cast_fp16)[name = string("op_6072_cast_fp16")]; + tensor concat_174 = const()[name = string("concat_174"), val = tensor([16, 128, 1024])]; + tensor reshape_27_cast_fp16 = reshape(shape = concat_174, x = var_6072_cast_fp16)[name = string("reshape_27_cast_fp16")]; + tensor concat_175 = const()[name = string("concat_175"), val = tensor([16, 1024, 128])]; + tensor reshape_28_cast_fp16 = reshape(shape = concat_175, x = x_155_cast_fp16)[name = string("reshape_28_cast_fp16")]; + bool matmul_9_transpose_x_0 = const()[name = string("matmul_9_transpose_x_0"), val = bool(false)]; + bool matmul_9_transpose_y_0 = const()[name = string("matmul_9_transpose_y_0"), val = bool(false)]; + tensor matmul_9_cast_fp16 = matmul(transpose_x = matmul_9_transpose_x_0, transpose_y = matmul_9_transpose_y_0, x = reshape_27_cast_fp16, y = reshape_28_cast_fp16)[name = string("matmul_9_cast_fp16")]; + tensor concat_179 = const()[name = string("concat_179"), val = tensor([1, 16, 128, 128])]; + tensor reshape_29_cast_fp16 = reshape(shape = concat_179, x = matmul_9_cast_fp16)[name = string("reshape_29_cast_fp16")]; + tensor var_6084_perm_0 = const()[name = string("op_6084_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_6103 = const()[name = string("op_6103"), val = tensor([1, 128, 2048])]; + tensor var_6084_cast_fp16 = transpose(perm = var_6084_perm_0, x = reshape_29_cast_fp16)[name = string("transpose_40")]; + tensor attn_output_95_cast_fp16 = reshape(shape = var_6103, x = var_6084_cast_fp16)[name = string("attn_output_95_cast_fp16")]; + tensor var_6108 = const()[name = string("op_6108"), val = tensor([0, 2, 1])]; + string var_6124_pad_type_0 = const()[name = string("op_6124_pad_type_0"), val = string("valid")]; + int32 var_6124_groups_0 = const()[name = string("op_6124_groups_0"), val = int32(1)]; + tensor var_6124_strides_0 = const()[name = string("op_6124_strides_0"), val = tensor([1])]; + tensor var_6124_pad_0 = const()[name = string("op_6124_pad_0"), val = tensor([0, 0])]; + tensor var_6124_dilations_0 = const()[name = string("op_6124_dilations_0"), val = tensor([1])]; + tensor squeeze_9_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(702512384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706706752))))[name = string("squeeze_9_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_6109_cast_fp16 = transpose(perm = var_6108, x = attn_output_95_cast_fp16)[name = string("transpose_39")]; + tensor var_6124_cast_fp16 = conv(dilations = var_6124_dilations_0, groups = var_6124_groups_0, pad = var_6124_pad_0, pad_type = var_6124_pad_type_0, strides = var_6124_strides_0, weight = squeeze_9_cast_fp16_to_fp32_to_fp16_palettized, x = var_6109_cast_fp16)[name = string("op_6124_cast_fp16")]; + tensor var_6128 = const()[name = string("op_6128"), val = tensor([0, 2, 1])]; + tensor attn_output_99_cast_fp16 = transpose(perm = var_6128, x = var_6124_cast_fp16)[name = string("transpose_38")]; + tensor hidden_states_99_cast_fp16 = add(x = hidden_states_91_cast_fp16, y = attn_output_99_cast_fp16)[name = string("hidden_states_99_cast_fp16")]; + int32 var_6141 = const()[name = string("op_6141"), val = int32(-1)]; + fp16 const_337_promoted_to_fp16 = const()[name = string("const_337_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6143_cast_fp16 = mul(x = hidden_states_99_cast_fp16, y = const_337_promoted_to_fp16)[name = string("op_6143_cast_fp16")]; + bool input_173_interleave_0 = const()[name = string("input_173_interleave_0"), val = bool(false)]; + tensor input_173_cast_fp16 = concat(axis = var_6141, interleave = input_173_interleave_0, values = (hidden_states_99_cast_fp16, var_6143_cast_fp16))[name = string("input_173_cast_fp16")]; + tensor normed_157_axes_0 = const()[name = string("normed_157_axes_0"), val = tensor([-1])]; + fp16 var_6138_to_fp16 = const()[name = string("op_6138_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_157_cast_fp16 = layer_norm(axes = normed_157_axes_0, epsilon = var_6138_to_fp16, x = input_173_cast_fp16)[name = string("normed_157_cast_fp16")]; + tensor normed_159_begin_0 = const()[name = string("normed_159_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_159_end_0 = const()[name = string("normed_159_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_159_end_mask_0 = const()[name = string("normed_159_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_159_cast_fp16 = slice_by_index(begin = normed_159_begin_0, end = normed_159_end_0, end_mask = normed_159_end_mask_0, x = normed_157_cast_fp16)[name = string("normed_159_cast_fp16")]; + tensor const_340_promoted_to_fp16 = const()[name = string("const_340_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706837888)))]; + tensor x_157_cast_fp16 = mul(x = normed_159_cast_fp16, y = const_340_promoted_to_fp16)[name = string("x_157_cast_fp16")]; + tensor var_6168 = const()[name = string("op_6168"), val = tensor([0, 2, 1])]; + tensor input_175_axes_0 = const()[name = string("input_175_axes_0"), val = tensor([2])]; + tensor var_6169 = transpose(perm = var_6168, x = x_157_cast_fp16)[name = string("transpose_37")]; + tensor input_175 = expand_dims(axes = input_175_axes_0, x = var_6169)[name = string("input_175")]; + string input_177_pad_type_0 = const()[name = string("input_177_pad_type_0"), val = string("valid")]; + tensor input_177_strides_0 = const()[name = string("input_177_strides_0"), val = tensor([1, 1])]; + tensor input_177_pad_0 = const()[name = string("input_177_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_177_dilations_0 = const()[name = string("input_177_dilations_0"), val = tensor([1, 1])]; + int32 input_177_groups_0 = const()[name = string("input_177_groups_0"), val = int32(1)]; + tensor input_177 = conv(dilations = input_177_dilations_0, groups = input_177_groups_0, pad = input_177_pad_0, pad_type = input_177_pad_type_0, strides = input_177_strides_0, weight = model_model_layers_23_mlp_gate_proj_weight_palettized, x = input_175)[name = string("input_177")]; + string b_19_pad_type_0 = const()[name = string("b_19_pad_type_0"), val = string("valid")]; + tensor b_19_strides_0 = const()[name = string("b_19_strides_0"), val = tensor([1, 1])]; + tensor b_19_pad_0 = const()[name = string("b_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_19_dilations_0 = const()[name = string("b_19_dilations_0"), val = tensor([1, 1])]; + int32 b_19_groups_0 = const()[name = string("b_19_groups_0"), val = int32(1)]; + tensor b_19 = conv(dilations = b_19_dilations_0, groups = b_19_groups_0, pad = b_19_pad_0, pad_type = b_19_pad_type_0, strides = b_19_strides_0, weight = model_model_layers_23_mlp_up_proj_weight_palettized, x = input_175)[name = string("b_19")]; + tensor c_19 = silu(x = input_177)[name = string("c_19")]; + tensor input_179 = mul(x = c_19, y = b_19)[name = string("input_179")]; + string e_19_pad_type_0 = const()[name = string("e_19_pad_type_0"), val = string("valid")]; + tensor e_19_strides_0 = const()[name = string("e_19_strides_0"), val = tensor([1, 1])]; + tensor e_19_pad_0 = const()[name = string("e_19_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_19_dilations_0 = const()[name = string("e_19_dilations_0"), val = tensor([1, 1])]; + int32 e_19_groups_0 = const()[name = string("e_19_groups_0"), val = int32(1)]; + tensor e_19 = conv(dilations = e_19_dilations_0, groups = e_19_groups_0, pad = e_19_pad_0, pad_type = e_19_pad_type_0, strides = e_19_strides_0, weight = model_model_layers_23_mlp_down_proj_weight_palettized, x = input_179)[name = string("e_19")]; + tensor var_6191_axes_0 = const()[name = string("op_6191_axes_0"), val = tensor([2])]; + tensor var_6191 = squeeze(axes = var_6191_axes_0, x = e_19)[name = string("op_6191")]; + tensor var_6192 = const()[name = string("op_6192"), val = tensor([0, 2, 1])]; + tensor var_6193 = transpose(perm = var_6192, x = var_6191)[name = string("transpose_36")]; + tensor hidden_states_101_cast_fp16 = add(x = hidden_states_99_cast_fp16, y = var_6193)[name = string("hidden_states_101_cast_fp16")]; + int32 var_6205 = const()[name = string("op_6205"), val = int32(-1)]; + fp16 const_341_promoted_to_fp16 = const()[name = string("const_341_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6207_cast_fp16 = mul(x = hidden_states_101_cast_fp16, y = const_341_promoted_to_fp16)[name = string("op_6207_cast_fp16")]; + bool input_181_interleave_0 = const()[name = string("input_181_interleave_0"), val = bool(false)]; + tensor input_181_cast_fp16 = concat(axis = var_6205, interleave = input_181_interleave_0, values = (hidden_states_101_cast_fp16, var_6207_cast_fp16))[name = string("input_181_cast_fp16")]; + tensor normed_161_axes_0 = const()[name = string("normed_161_axes_0"), val = tensor([-1])]; + fp16 var_6202_to_fp16 = const()[name = string("op_6202_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_161_cast_fp16 = layer_norm(axes = normed_161_axes_0, epsilon = var_6202_to_fp16, x = input_181_cast_fp16)[name = string("normed_161_cast_fp16")]; + tensor normed_163_begin_0 = const()[name = string("normed_163_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_163_end_0 = const()[name = string("normed_163_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_163_end_mask_0 = const()[name = string("normed_163_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_163_cast_fp16 = slice_by_index(begin = normed_163_begin_0, end = normed_163_end_0, end_mask = normed_163_end_mask_0, x = normed_161_cast_fp16)[name = string("normed_163_cast_fp16")]; + tensor const_344_promoted_to_fp16 = const()[name = string("const_344_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706842048)))]; + tensor hidden_states_103_cast_fp16 = mul(x = normed_163_cast_fp16, y = const_344_promoted_to_fp16)[name = string("hidden_states_103_cast_fp16")]; + tensor var_6230 = const()[name = string("op_6230"), val = tensor([0, 2, 1])]; + tensor var_6233_axes_0 = const()[name = string("op_6233_axes_0"), val = tensor([2])]; + tensor var_6231_cast_fp16 = transpose(perm = var_6230, x = hidden_states_103_cast_fp16)[name = string("transpose_35")]; + tensor var_6233_cast_fp16 = expand_dims(axes = var_6233_axes_0, x = var_6231_cast_fp16)[name = string("op_6233_cast_fp16")]; + string query_states_81_pad_type_0 = const()[name = string("query_states_81_pad_type_0"), val = string("valid")]; + tensor query_states_81_strides_0 = const()[name = string("query_states_81_strides_0"), val = tensor([1, 1])]; + tensor query_states_81_pad_0 = const()[name = string("query_states_81_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_81_dilations_0 = const()[name = string("query_states_81_dilations_0"), val = tensor([1, 1])]; + int32 query_states_81_groups_0 = const()[name = string("query_states_81_groups_0"), val = int32(1)]; + tensor query_states_81 = conv(dilations = query_states_81_dilations_0, groups = query_states_81_groups_0, pad = query_states_81_pad_0, pad_type = query_states_81_pad_type_0, strides = query_states_81_strides_0, weight = model_model_layers_24_self_attn_q_proj_weight_palettized, x = var_6233_cast_fp16)[name = string("query_states_81")]; + string key_states_101_pad_type_0 = const()[name = string("key_states_101_pad_type_0"), val = string("valid")]; + tensor key_states_101_strides_0 = const()[name = string("key_states_101_strides_0"), val = tensor([1, 1])]; + tensor key_states_101_pad_0 = const()[name = string("key_states_101_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_101_dilations_0 = const()[name = string("key_states_101_dilations_0"), val = tensor([1, 1])]; + int32 key_states_101_groups_0 = const()[name = string("key_states_101_groups_0"), val = int32(1)]; + tensor key_states_101 = conv(dilations = key_states_101_dilations_0, groups = key_states_101_groups_0, pad = key_states_101_pad_0, pad_type = key_states_101_pad_type_0, strides = key_states_101_strides_0, weight = model_model_layers_24_self_attn_k_proj_weight_palettized, x = var_6233_cast_fp16)[name = string("key_states_101")]; + string value_states_81_pad_type_0 = const()[name = string("value_states_81_pad_type_0"), val = string("valid")]; + tensor value_states_81_strides_0 = const()[name = string("value_states_81_strides_0"), val = tensor([1, 1])]; + tensor value_states_81_pad_0 = const()[name = string("value_states_81_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_81_dilations_0 = const()[name = string("value_states_81_dilations_0"), val = tensor([1, 1])]; + int32 value_states_81_groups_0 = const()[name = string("value_states_81_groups_0"), val = int32(1)]; + tensor value_states_81 = conv(dilations = value_states_81_dilations_0, groups = value_states_81_groups_0, pad = value_states_81_pad_0, pad_type = value_states_81_pad_type_0, strides = value_states_81_strides_0, weight = model_model_layers_24_self_attn_v_proj_weight_palettized, x = var_6233_cast_fp16)[name = string("value_states_81")]; + tensor var_6275 = const()[name = string("op_6275"), val = tensor([1, 16, 128, 128])]; + tensor var_6276 = reshape(shape = var_6275, x = query_states_81)[name = string("op_6276")]; + tensor var_6281 = const()[name = string("op_6281"), val = tensor([0, 1, 3, 2])]; + tensor var_6286 = const()[name = string("op_6286"), val = tensor([1, 8, 128, 128])]; + tensor var_6287 = reshape(shape = var_6286, x = key_states_101)[name = string("op_6287")]; + tensor var_6292 = const()[name = string("op_6292"), val = tensor([0, 1, 3, 2])]; + tensor var_6297 = const()[name = string("op_6297"), val = tensor([1, 8, 128, 128])]; + tensor var_6298 = reshape(shape = var_6297, x = value_states_81)[name = string("op_6298")]; + tensor var_6303 = const()[name = string("op_6303"), val = tensor([0, 1, 3, 2])]; + int32 var_6314 = const()[name = string("op_6314"), val = int32(-1)]; + fp16 const_346_promoted = const()[name = string("const_346_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_105 = transpose(perm = var_6281, x = var_6276)[name = string("transpose_34")]; + tensor var_6316 = mul(x = hidden_states_105, y = const_346_promoted)[name = string("op_6316")]; + bool input_185_interleave_0 = const()[name = string("input_185_interleave_0"), val = bool(false)]; + tensor input_185 = concat(axis = var_6314, interleave = input_185_interleave_0, values = (hidden_states_105, var_6316))[name = string("input_185")]; + tensor normed_165_axes_0 = const()[name = string("normed_165_axes_0"), val = tensor([-1])]; + fp16 var_6311_to_fp16 = const()[name = string("op_6311_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_165_cast_fp16 = layer_norm(axes = normed_165_axes_0, epsilon = var_6311_to_fp16, x = input_185)[name = string("normed_165_cast_fp16")]; + tensor normed_167_begin_0 = const()[name = string("normed_167_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_167_end_0 = const()[name = string("normed_167_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_167_end_mask_0 = const()[name = string("normed_167_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_167 = slice_by_index(begin = normed_167_begin_0, end = normed_167_end_0, end_mask = normed_167_end_mask_0, x = normed_165_cast_fp16)[name = string("normed_167")]; + tensor const_349 = const()[name = string("const_349"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706846208)))]; + tensor q_21 = mul(x = normed_167, y = const_349)[name = string("q_21")]; + int32 var_6339 = const()[name = string("op_6339"), val = int32(-1)]; + fp16 const_350_promoted = const()[name = string("const_350_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_107 = transpose(perm = var_6292, x = var_6287)[name = string("transpose_33")]; + tensor var_6341 = mul(x = hidden_states_107, y = const_350_promoted)[name = string("op_6341")]; + bool input_187_interleave_0 = const()[name = string("input_187_interleave_0"), val = bool(false)]; + tensor input_187 = concat(axis = var_6339, interleave = input_187_interleave_0, values = (hidden_states_107, var_6341))[name = string("input_187")]; + tensor normed_169_axes_0 = const()[name = string("normed_169_axes_0"), val = tensor([-1])]; + fp16 var_6336_to_fp16 = const()[name = string("op_6336_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_169_cast_fp16 = layer_norm(axes = normed_169_axes_0, epsilon = var_6336_to_fp16, x = input_187)[name = string("normed_169_cast_fp16")]; + tensor normed_171_begin_0 = const()[name = string("normed_171_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_171_end_0 = const()[name = string("normed_171_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_171_end_mask_0 = const()[name = string("normed_171_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_171 = slice_by_index(begin = normed_171_begin_0, end = normed_171_end_0, end_mask = normed_171_end_mask_0, x = normed_169_cast_fp16)[name = string("normed_171")]; + tensor const_353 = const()[name = string("const_353"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706846528)))]; + tensor k_21 = mul(x = normed_171, y = const_353)[name = string("k_21")]; + tensor var_6367 = mul(x = q_21, y = cos_5)[name = string("op_6367")]; + tensor x1_41_begin_0 = const()[name = string("x1_41_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_41_end_0 = const()[name = string("x1_41_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_41_end_mask_0 = const()[name = string("x1_41_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_41 = slice_by_index(begin = x1_41_begin_0, end = x1_41_end_0, end_mask = x1_41_end_mask_0, x = q_21)[name = string("x1_41")]; + tensor x2_41_begin_0 = const()[name = string("x2_41_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_41_end_0 = const()[name = string("x2_41_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_41_end_mask_0 = const()[name = string("x2_41_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_41 = slice_by_index(begin = x2_41_begin_0, end = x2_41_end_0, end_mask = x2_41_end_mask_0, x = q_21)[name = string("x2_41")]; + fp16 const_356_promoted = const()[name = string("const_356_promoted"), val = fp16(-0x1p+0)]; + tensor var_6388 = mul(x = x2_41, y = const_356_promoted)[name = string("op_6388")]; + int32 var_6390 = const()[name = string("op_6390"), val = int32(-1)]; + bool var_6391_interleave_0 = const()[name = string("op_6391_interleave_0"), val = bool(false)]; + tensor var_6391 = concat(axis = var_6390, interleave = var_6391_interleave_0, values = (var_6388, x1_41))[name = string("op_6391")]; + tensor var_6392 = mul(x = var_6391, y = sin_5)[name = string("op_6392")]; + tensor query_states_83 = add(x = var_6367, y = var_6392)[name = string("query_states_83")]; + tensor var_6395 = mul(x = k_21, y = cos_5)[name = string("op_6395")]; + tensor x1_43_begin_0 = const()[name = string("x1_43_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_43_end_0 = const()[name = string("x1_43_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_43_end_mask_0 = const()[name = string("x1_43_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_43 = slice_by_index(begin = x1_43_begin_0, end = x1_43_end_0, end_mask = x1_43_end_mask_0, x = k_21)[name = string("x1_43")]; + tensor x2_43_begin_0 = const()[name = string("x2_43_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_43_end_0 = const()[name = string("x2_43_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_43_end_mask_0 = const()[name = string("x2_43_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_43 = slice_by_index(begin = x2_43_begin_0, end = x2_43_end_0, end_mask = x2_43_end_mask_0, x = k_21)[name = string("x2_43")]; + fp16 const_359_promoted = const()[name = string("const_359_promoted"), val = fp16(-0x1p+0)]; + tensor var_6416 = mul(x = x2_43, y = const_359_promoted)[name = string("op_6416")]; + int32 var_6418 = const()[name = string("op_6418"), val = int32(-1)]; + bool var_6419_interleave_0 = const()[name = string("op_6419_interleave_0"), val = bool(false)]; + tensor var_6419 = concat(axis = var_6418, interleave = var_6419_interleave_0, values = (var_6416, x1_43))[name = string("op_6419")]; + tensor var_6420 = mul(x = var_6419, y = sin_5)[name = string("op_6420")]; + tensor key_states_103 = add(x = var_6395, y = var_6420)[name = string("key_states_103")]; + tensor expand_dims_120 = const()[name = string("expand_dims_120"), val = tensor([24])]; + tensor expand_dims_121 = const()[name = string("expand_dims_121"), val = tensor([0])]; + tensor expand_dims_123 = const()[name = string("expand_dims_123"), val = tensor([0])]; + tensor expand_dims_124 = const()[name = string("expand_dims_124"), val = tensor([25])]; + int32 concat_182_axis_0 = const()[name = string("concat_182_axis_0"), val = int32(0)]; + bool concat_182_interleave_0 = const()[name = string("concat_182_interleave_0"), val = bool(false)]; + tensor concat_182 = concat(axis = concat_182_axis_0, interleave = concat_182_interleave_0, values = (expand_dims_120, expand_dims_121, current_pos, expand_dims_123))[name = string("concat_182")]; + tensor concat_183_values1_0 = const()[name = string("concat_183_values1_0"), val = tensor([0])]; + tensor concat_183_values3_0 = const()[name = string("concat_183_values3_0"), val = tensor([0])]; + int32 concat_183_axis_0 = const()[name = string("concat_183_axis_0"), val = int32(0)]; + bool concat_183_interleave_0 = const()[name = string("concat_183_interleave_0"), val = bool(false)]; + tensor concat_183 = concat(axis = concat_183_axis_0, interleave = concat_183_interleave_0, values = (expand_dims_124, concat_183_values1_0, var_1042, concat_183_values3_0))[name = string("concat_183")]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_182, begin_mask = model_model_kv_cache_0_internal_tensor_assign_21_begin_mask_0, end = concat_183, end_mask = model_model_kv_cache_0_internal_tensor_assign_21_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_21_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_21_stride_0, update = key_states_103, x = coreml_update_state_47)[name = string("model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_21_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_104_write_state")]; + tensor coreml_update_state_48 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_104")]; + tensor expand_dims_126 = const()[name = string("expand_dims_126"), val = tensor([52])]; + tensor expand_dims_127 = const()[name = string("expand_dims_127"), val = tensor([0])]; + tensor expand_dims_129 = const()[name = string("expand_dims_129"), val = tensor([0])]; + tensor expand_dims_130 = const()[name = string("expand_dims_130"), val = tensor([53])]; + int32 concat_186_axis_0 = const()[name = string("concat_186_axis_0"), val = int32(0)]; + bool concat_186_interleave_0 = const()[name = string("concat_186_interleave_0"), val = bool(false)]; + tensor concat_186 = concat(axis = concat_186_axis_0, interleave = concat_186_interleave_0, values = (expand_dims_126, expand_dims_127, current_pos, expand_dims_129))[name = string("concat_186")]; + tensor concat_187_values1_0 = const()[name = string("concat_187_values1_0"), val = tensor([0])]; + tensor concat_187_values3_0 = const()[name = string("concat_187_values3_0"), val = tensor([0])]; + int32 concat_187_axis_0 = const()[name = string("concat_187_axis_0"), val = int32(0)]; + bool concat_187_interleave_0 = const()[name = string("concat_187_interleave_0"), val = bool(false)]; + tensor concat_187 = concat(axis = concat_187_axis_0, interleave = concat_187_interleave_0, values = (expand_dims_130, concat_187_values1_0, var_1042, concat_187_values3_0))[name = string("concat_187")]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_83 = transpose(perm = var_6303, x = var_6298)[name = string("transpose_32")]; + tensor model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_186, begin_mask = model_model_kv_cache_0_internal_tensor_assign_22_begin_mask_0, end = concat_187, end_mask = model_model_kv_cache_0_internal_tensor_assign_22_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_22_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_22_stride_0, update = value_states_83, x = coreml_update_state_48)[name = string("model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_22_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_105_write_state")]; + tensor coreml_update_state_49 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_105")]; + tensor var_6491_begin_0 = const()[name = string("op_6491_begin_0"), val = tensor([24, 0, 0, 0])]; + tensor var_6491_end_0 = const()[name = string("op_6491_end_0"), val = tensor([25, 8, 1024, 128])]; + tensor var_6491_end_mask_0 = const()[name = string("op_6491_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_6491_cast_fp16 = slice_by_index(begin = var_6491_begin_0, end = var_6491_end_0, end_mask = var_6491_end_mask_0, x = coreml_update_state_49)[name = string("op_6491_cast_fp16")]; + tensor K_layer_cache_21_axes_0 = const()[name = string("K_layer_cache_21_axes_0"), val = tensor([0])]; + tensor K_layer_cache_21_cast_fp16 = squeeze(axes = K_layer_cache_21_axes_0, x = var_6491_cast_fp16)[name = string("K_layer_cache_21_cast_fp16")]; + tensor var_6498_begin_0 = const()[name = string("op_6498_begin_0"), val = tensor([52, 0, 0, 0])]; + tensor var_6498_end_0 = const()[name = string("op_6498_end_0"), val = tensor([53, 8, 1024, 128])]; + tensor var_6498_end_mask_0 = const()[name = string("op_6498_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_6498_cast_fp16 = slice_by_index(begin = var_6498_begin_0, end = var_6498_end_0, end_mask = var_6498_end_mask_0, x = coreml_update_state_49)[name = string("op_6498_cast_fp16")]; + tensor V_layer_cache_21_axes_0 = const()[name = string("V_layer_cache_21_axes_0"), val = tensor([0])]; + tensor V_layer_cache_21_cast_fp16 = squeeze(axes = V_layer_cache_21_axes_0, x = var_6498_cast_fp16)[name = string("V_layer_cache_21_cast_fp16")]; + tensor x_163_axes_0 = const()[name = string("x_163_axes_0"), val = tensor([1])]; + tensor x_163_cast_fp16 = expand_dims(axes = x_163_axes_0, x = K_layer_cache_21_cast_fp16)[name = string("x_163_cast_fp16")]; + tensor var_6527 = const()[name = string("op_6527"), val = tensor([1, 2, 1, 1])]; + tensor x_165_cast_fp16 = tile(reps = var_6527, x = x_163_cast_fp16)[name = string("x_165_cast_fp16")]; + tensor var_6539 = const()[name = string("op_6539"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_107_cast_fp16 = reshape(shape = var_6539, x = x_165_cast_fp16)[name = string("key_states_107_cast_fp16")]; + tensor x_169_axes_0 = const()[name = string("x_169_axes_0"), val = tensor([1])]; + tensor x_169_cast_fp16 = expand_dims(axes = x_169_axes_0, x = V_layer_cache_21_cast_fp16)[name = string("x_169_cast_fp16")]; + tensor var_6547 = const()[name = string("op_6547"), val = tensor([1, 2, 1, 1])]; + tensor x_171_cast_fp16 = tile(reps = var_6547, x = x_169_cast_fp16)[name = string("x_171_cast_fp16")]; + bool var_6574_transpose_x_0 = const()[name = string("op_6574_transpose_x_0"), val = bool(false)]; + bool var_6574_transpose_y_0 = const()[name = string("op_6574_transpose_y_0"), val = bool(true)]; + tensor var_6574 = matmul(transpose_x = var_6574_transpose_x_0, transpose_y = var_6574_transpose_y_0, x = query_states_83, y = key_states_107_cast_fp16)[name = string("op_6574")]; + fp16 var_6575_to_fp16 = const()[name = string("op_6575_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_41_cast_fp16 = mul(x = var_6574, y = var_6575_to_fp16)[name = string("attn_weights_41_cast_fp16")]; + tensor attn_weights_43_cast_fp16 = add(x = attn_weights_41_cast_fp16, y = causal_mask)[name = string("attn_weights_43_cast_fp16")]; + int32 var_6610 = const()[name = string("op_6610"), val = int32(-1)]; + tensor var_6612_cast_fp16 = softmax(axis = var_6610, x = attn_weights_43_cast_fp16)[name = string("op_6612_cast_fp16")]; + tensor concat_192 = const()[name = string("concat_192"), val = tensor([16, 128, 1024])]; + tensor reshape_30_cast_fp16 = reshape(shape = concat_192, x = var_6612_cast_fp16)[name = string("reshape_30_cast_fp16")]; + tensor concat_193 = const()[name = string("concat_193"), val = tensor([16, 1024, 128])]; + tensor reshape_31_cast_fp16 = reshape(shape = concat_193, x = x_171_cast_fp16)[name = string("reshape_31_cast_fp16")]; + bool matmul_10_transpose_x_0 = const()[name = string("matmul_10_transpose_x_0"), val = bool(false)]; + bool matmul_10_transpose_y_0 = const()[name = string("matmul_10_transpose_y_0"), val = bool(false)]; + tensor matmul_10_cast_fp16 = matmul(transpose_x = matmul_10_transpose_x_0, transpose_y = matmul_10_transpose_y_0, x = reshape_30_cast_fp16, y = reshape_31_cast_fp16)[name = string("matmul_10_cast_fp16")]; + tensor concat_197 = const()[name = string("concat_197"), val = tensor([1, 16, 128, 128])]; + tensor reshape_32_cast_fp16 = reshape(shape = concat_197, x = matmul_10_cast_fp16)[name = string("reshape_32_cast_fp16")]; + tensor var_6624_perm_0 = const()[name = string("op_6624_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_6643 = const()[name = string("op_6643"), val = tensor([1, 128, 2048])]; + tensor var_6624_cast_fp16 = transpose(perm = var_6624_perm_0, x = reshape_32_cast_fp16)[name = string("transpose_31")]; + tensor attn_output_105_cast_fp16 = reshape(shape = var_6643, x = var_6624_cast_fp16)[name = string("attn_output_105_cast_fp16")]; + tensor var_6648 = const()[name = string("op_6648"), val = tensor([0, 2, 1])]; + string var_6664_pad_type_0 = const()[name = string("op_6664_pad_type_0"), val = string("valid")]; + int32 var_6664_groups_0 = const()[name = string("op_6664_groups_0"), val = int32(1)]; + tensor var_6664_strides_0 = const()[name = string("op_6664_strides_0"), val = tensor([1])]; + tensor var_6664_pad_0 = const()[name = string("op_6664_pad_0"), val = tensor([0, 0])]; + tensor var_6664_dilations_0 = const()[name = string("op_6664_dilations_0"), val = tensor([1])]; + tensor squeeze_10_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(706846848))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711041216))))[name = string("squeeze_10_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_6649_cast_fp16 = transpose(perm = var_6648, x = attn_output_105_cast_fp16)[name = string("transpose_30")]; + tensor var_6664_cast_fp16 = conv(dilations = var_6664_dilations_0, groups = var_6664_groups_0, pad = var_6664_pad_0, pad_type = var_6664_pad_type_0, strides = var_6664_strides_0, weight = squeeze_10_cast_fp16_to_fp32_to_fp16_palettized, x = var_6649_cast_fp16)[name = string("op_6664_cast_fp16")]; + tensor var_6668 = const()[name = string("op_6668"), val = tensor([0, 2, 1])]; + tensor attn_output_109_cast_fp16 = transpose(perm = var_6668, x = var_6664_cast_fp16)[name = string("transpose_29")]; + tensor hidden_states_109_cast_fp16 = add(x = hidden_states_101_cast_fp16, y = attn_output_109_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; + int32 var_6681 = const()[name = string("op_6681"), val = int32(-1)]; + fp16 const_371_promoted_to_fp16 = const()[name = string("const_371_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6683_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = const_371_promoted_to_fp16)[name = string("op_6683_cast_fp16")]; + bool input_191_interleave_0 = const()[name = string("input_191_interleave_0"), val = bool(false)]; + tensor input_191_cast_fp16 = concat(axis = var_6681, interleave = input_191_interleave_0, values = (hidden_states_109_cast_fp16, var_6683_cast_fp16))[name = string("input_191_cast_fp16")]; + tensor normed_173_axes_0 = const()[name = string("normed_173_axes_0"), val = tensor([-1])]; + fp16 var_6678_to_fp16 = const()[name = string("op_6678_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_173_cast_fp16 = layer_norm(axes = normed_173_axes_0, epsilon = var_6678_to_fp16, x = input_191_cast_fp16)[name = string("normed_173_cast_fp16")]; + tensor normed_175_begin_0 = const()[name = string("normed_175_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_175_end_0 = const()[name = string("normed_175_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_175_end_mask_0 = const()[name = string("normed_175_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_175_cast_fp16 = slice_by_index(begin = normed_175_begin_0, end = normed_175_end_0, end_mask = normed_175_end_mask_0, x = normed_173_cast_fp16)[name = string("normed_175_cast_fp16")]; + tensor const_374_promoted_to_fp16 = const()[name = string("const_374_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711172352)))]; + tensor x_173_cast_fp16 = mul(x = normed_175_cast_fp16, y = const_374_promoted_to_fp16)[name = string("x_173_cast_fp16")]; + tensor var_6708 = const()[name = string("op_6708"), val = tensor([0, 2, 1])]; + tensor input_193_axes_0 = const()[name = string("input_193_axes_0"), val = tensor([2])]; + tensor var_6709 = transpose(perm = var_6708, x = x_173_cast_fp16)[name = string("transpose_28")]; + tensor input_193 = expand_dims(axes = input_193_axes_0, x = var_6709)[name = string("input_193")]; + string input_195_pad_type_0 = const()[name = string("input_195_pad_type_0"), val = string("valid")]; + tensor input_195_strides_0 = const()[name = string("input_195_strides_0"), val = tensor([1, 1])]; + tensor input_195_pad_0 = const()[name = string("input_195_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_195_dilations_0 = const()[name = string("input_195_dilations_0"), val = tensor([1, 1])]; + int32 input_195_groups_0 = const()[name = string("input_195_groups_0"), val = int32(1)]; + tensor input_195 = conv(dilations = input_195_dilations_0, groups = input_195_groups_0, pad = input_195_pad_0, pad_type = input_195_pad_type_0, strides = input_195_strides_0, weight = model_model_layers_24_mlp_gate_proj_weight_palettized, x = input_193)[name = string("input_195")]; + string b_21_pad_type_0 = const()[name = string("b_21_pad_type_0"), val = string("valid")]; + tensor b_21_strides_0 = const()[name = string("b_21_strides_0"), val = tensor([1, 1])]; + tensor b_21_pad_0 = const()[name = string("b_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_21_dilations_0 = const()[name = string("b_21_dilations_0"), val = tensor([1, 1])]; + int32 b_21_groups_0 = const()[name = string("b_21_groups_0"), val = int32(1)]; + tensor b_21 = conv(dilations = b_21_dilations_0, groups = b_21_groups_0, pad = b_21_pad_0, pad_type = b_21_pad_type_0, strides = b_21_strides_0, weight = model_model_layers_24_mlp_up_proj_weight_palettized, x = input_193)[name = string("b_21")]; + tensor c_21 = silu(x = input_195)[name = string("c_21")]; + tensor input_197 = mul(x = c_21, y = b_21)[name = string("input_197")]; + string e_21_pad_type_0 = const()[name = string("e_21_pad_type_0"), val = string("valid")]; + tensor e_21_strides_0 = const()[name = string("e_21_strides_0"), val = tensor([1, 1])]; + tensor e_21_pad_0 = const()[name = string("e_21_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_21_dilations_0 = const()[name = string("e_21_dilations_0"), val = tensor([1, 1])]; + int32 e_21_groups_0 = const()[name = string("e_21_groups_0"), val = int32(1)]; + tensor e_21 = conv(dilations = e_21_dilations_0, groups = e_21_groups_0, pad = e_21_pad_0, pad_type = e_21_pad_type_0, strides = e_21_strides_0, weight = model_model_layers_24_mlp_down_proj_weight_palettized, x = input_197)[name = string("e_21")]; + tensor var_6731_axes_0 = const()[name = string("op_6731_axes_0"), val = tensor([2])]; + tensor var_6731 = squeeze(axes = var_6731_axes_0, x = e_21)[name = string("op_6731")]; + tensor var_6732 = const()[name = string("op_6732"), val = tensor([0, 2, 1])]; + tensor var_6733 = transpose(perm = var_6732, x = var_6731)[name = string("transpose_27")]; + tensor hidden_states_111_cast_fp16 = add(x = hidden_states_109_cast_fp16, y = var_6733)[name = string("hidden_states_111_cast_fp16")]; + int32 var_6745 = const()[name = string("op_6745"), val = int32(-1)]; + fp16 const_375_promoted_to_fp16 = const()[name = string("const_375_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_6747_cast_fp16 = mul(x = hidden_states_111_cast_fp16, y = const_375_promoted_to_fp16)[name = string("op_6747_cast_fp16")]; + bool input_199_interleave_0 = const()[name = string("input_199_interleave_0"), val = bool(false)]; + tensor input_199_cast_fp16 = concat(axis = var_6745, interleave = input_199_interleave_0, values = (hidden_states_111_cast_fp16, var_6747_cast_fp16))[name = string("input_199_cast_fp16")]; + tensor normed_177_axes_0 = const()[name = string("normed_177_axes_0"), val = tensor([-1])]; + fp16 var_6742_to_fp16 = const()[name = string("op_6742_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_177_cast_fp16 = layer_norm(axes = normed_177_axes_0, epsilon = var_6742_to_fp16, x = input_199_cast_fp16)[name = string("normed_177_cast_fp16")]; + tensor normed_179_begin_0 = const()[name = string("normed_179_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_179_end_0 = const()[name = string("normed_179_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_179_end_mask_0 = const()[name = string("normed_179_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_179_cast_fp16 = slice_by_index(begin = normed_179_begin_0, end = normed_179_end_0, end_mask = normed_179_end_mask_0, x = normed_177_cast_fp16)[name = string("normed_179_cast_fp16")]; + tensor const_378_promoted_to_fp16 = const()[name = string("const_378_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711176512)))]; + tensor hidden_states_113_cast_fp16 = mul(x = normed_179_cast_fp16, y = const_378_promoted_to_fp16)[name = string("hidden_states_113_cast_fp16")]; + tensor var_6770 = const()[name = string("op_6770"), val = tensor([0, 2, 1])]; + tensor var_6773_axes_0 = const()[name = string("op_6773_axes_0"), val = tensor([2])]; + tensor var_6771_cast_fp16 = transpose(perm = var_6770, x = hidden_states_113_cast_fp16)[name = string("transpose_26")]; + tensor var_6773_cast_fp16 = expand_dims(axes = var_6773_axes_0, x = var_6771_cast_fp16)[name = string("op_6773_cast_fp16")]; + string query_states_89_pad_type_0 = const()[name = string("query_states_89_pad_type_0"), val = string("valid")]; + tensor query_states_89_strides_0 = const()[name = string("query_states_89_strides_0"), val = tensor([1, 1])]; + tensor query_states_89_pad_0 = const()[name = string("query_states_89_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_89_dilations_0 = const()[name = string("query_states_89_dilations_0"), val = tensor([1, 1])]; + int32 query_states_89_groups_0 = const()[name = string("query_states_89_groups_0"), val = int32(1)]; + tensor query_states_89 = conv(dilations = query_states_89_dilations_0, groups = query_states_89_groups_0, pad = query_states_89_pad_0, pad_type = query_states_89_pad_type_0, strides = query_states_89_strides_0, weight = model_model_layers_25_self_attn_q_proj_weight_palettized, x = var_6773_cast_fp16)[name = string("query_states_89")]; + string key_states_111_pad_type_0 = const()[name = string("key_states_111_pad_type_0"), val = string("valid")]; + tensor key_states_111_strides_0 = const()[name = string("key_states_111_strides_0"), val = tensor([1, 1])]; + tensor key_states_111_pad_0 = const()[name = string("key_states_111_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_111_dilations_0 = const()[name = string("key_states_111_dilations_0"), val = tensor([1, 1])]; + int32 key_states_111_groups_0 = const()[name = string("key_states_111_groups_0"), val = int32(1)]; + tensor key_states_111 = conv(dilations = key_states_111_dilations_0, groups = key_states_111_groups_0, pad = key_states_111_pad_0, pad_type = key_states_111_pad_type_0, strides = key_states_111_strides_0, weight = model_model_layers_25_self_attn_k_proj_weight_palettized, x = var_6773_cast_fp16)[name = string("key_states_111")]; + string value_states_89_pad_type_0 = const()[name = string("value_states_89_pad_type_0"), val = string("valid")]; + tensor value_states_89_strides_0 = const()[name = string("value_states_89_strides_0"), val = tensor([1, 1])]; + tensor value_states_89_pad_0 = const()[name = string("value_states_89_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_89_dilations_0 = const()[name = string("value_states_89_dilations_0"), val = tensor([1, 1])]; + int32 value_states_89_groups_0 = const()[name = string("value_states_89_groups_0"), val = int32(1)]; + tensor value_states_89 = conv(dilations = value_states_89_dilations_0, groups = value_states_89_groups_0, pad = value_states_89_pad_0, pad_type = value_states_89_pad_type_0, strides = value_states_89_strides_0, weight = model_model_layers_25_self_attn_v_proj_weight_palettized, x = var_6773_cast_fp16)[name = string("value_states_89")]; + tensor var_6815 = const()[name = string("op_6815"), val = tensor([1, 16, 128, 128])]; + tensor var_6816 = reshape(shape = var_6815, x = query_states_89)[name = string("op_6816")]; + tensor var_6821 = const()[name = string("op_6821"), val = tensor([0, 1, 3, 2])]; + tensor var_6826 = const()[name = string("op_6826"), val = tensor([1, 8, 128, 128])]; + tensor var_6827 = reshape(shape = var_6826, x = key_states_111)[name = string("op_6827")]; + tensor var_6832 = const()[name = string("op_6832"), val = tensor([0, 1, 3, 2])]; + tensor var_6837 = const()[name = string("op_6837"), val = tensor([1, 8, 128, 128])]; + tensor var_6838 = reshape(shape = var_6837, x = value_states_89)[name = string("op_6838")]; + tensor var_6843 = const()[name = string("op_6843"), val = tensor([0, 1, 3, 2])]; + int32 var_6854 = const()[name = string("op_6854"), val = int32(-1)]; + fp16 const_380_promoted = const()[name = string("const_380_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_115 = transpose(perm = var_6821, x = var_6816)[name = string("transpose_25")]; + tensor var_6856 = mul(x = hidden_states_115, y = const_380_promoted)[name = string("op_6856")]; + bool input_203_interleave_0 = const()[name = string("input_203_interleave_0"), val = bool(false)]; + tensor input_203 = concat(axis = var_6854, interleave = input_203_interleave_0, values = (hidden_states_115, var_6856))[name = string("input_203")]; + tensor normed_181_axes_0 = const()[name = string("normed_181_axes_0"), val = tensor([-1])]; + fp16 var_6851_to_fp16 = const()[name = string("op_6851_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_181_cast_fp16 = layer_norm(axes = normed_181_axes_0, epsilon = var_6851_to_fp16, x = input_203)[name = string("normed_181_cast_fp16")]; + tensor normed_183_begin_0 = const()[name = string("normed_183_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_183_end_0 = const()[name = string("normed_183_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_183_end_mask_0 = const()[name = string("normed_183_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_183 = slice_by_index(begin = normed_183_begin_0, end = normed_183_end_0, end_mask = normed_183_end_mask_0, x = normed_181_cast_fp16)[name = string("normed_183")]; + tensor const_383 = const()[name = string("const_383"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711180672)))]; + tensor q_23 = mul(x = normed_183, y = const_383)[name = string("q_23")]; + int32 var_6879 = const()[name = string("op_6879"), val = int32(-1)]; + fp16 const_384_promoted = const()[name = string("const_384_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_117 = transpose(perm = var_6832, x = var_6827)[name = string("transpose_24")]; + tensor var_6881 = mul(x = hidden_states_117, y = const_384_promoted)[name = string("op_6881")]; + bool input_205_interleave_0 = const()[name = string("input_205_interleave_0"), val = bool(false)]; + tensor input_205 = concat(axis = var_6879, interleave = input_205_interleave_0, values = (hidden_states_117, var_6881))[name = string("input_205")]; + tensor normed_185_axes_0 = const()[name = string("normed_185_axes_0"), val = tensor([-1])]; + fp16 var_6876_to_fp16 = const()[name = string("op_6876_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_185_cast_fp16 = layer_norm(axes = normed_185_axes_0, epsilon = var_6876_to_fp16, x = input_205)[name = string("normed_185_cast_fp16")]; + tensor normed_187_begin_0 = const()[name = string("normed_187_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_187_end_0 = const()[name = string("normed_187_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_187_end_mask_0 = const()[name = string("normed_187_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_187 = slice_by_index(begin = normed_187_begin_0, end = normed_187_end_0, end_mask = normed_187_end_mask_0, x = normed_185_cast_fp16)[name = string("normed_187")]; + tensor const_387 = const()[name = string("const_387"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711180992)))]; + tensor k_23 = mul(x = normed_187, y = const_387)[name = string("k_23")]; + tensor var_6907 = mul(x = q_23, y = cos_5)[name = string("op_6907")]; + tensor x1_45_begin_0 = const()[name = string("x1_45_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_45_end_0 = const()[name = string("x1_45_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_45_end_mask_0 = const()[name = string("x1_45_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_45 = slice_by_index(begin = x1_45_begin_0, end = x1_45_end_0, end_mask = x1_45_end_mask_0, x = q_23)[name = string("x1_45")]; + tensor x2_45_begin_0 = const()[name = string("x2_45_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_45_end_0 = const()[name = string("x2_45_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_45_end_mask_0 = const()[name = string("x2_45_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_45 = slice_by_index(begin = x2_45_begin_0, end = x2_45_end_0, end_mask = x2_45_end_mask_0, x = q_23)[name = string("x2_45")]; + fp16 const_390_promoted = const()[name = string("const_390_promoted"), val = fp16(-0x1p+0)]; + tensor var_6928 = mul(x = x2_45, y = const_390_promoted)[name = string("op_6928")]; + int32 var_6930 = const()[name = string("op_6930"), val = int32(-1)]; + bool var_6931_interleave_0 = const()[name = string("op_6931_interleave_0"), val = bool(false)]; + tensor var_6931 = concat(axis = var_6930, interleave = var_6931_interleave_0, values = (var_6928, x1_45))[name = string("op_6931")]; + tensor var_6932 = mul(x = var_6931, y = sin_5)[name = string("op_6932")]; + tensor query_states_91 = add(x = var_6907, y = var_6932)[name = string("query_states_91")]; + tensor var_6935 = mul(x = k_23, y = cos_5)[name = string("op_6935")]; + tensor x1_47_begin_0 = const()[name = string("x1_47_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_47_end_0 = const()[name = string("x1_47_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_47_end_mask_0 = const()[name = string("x1_47_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_47 = slice_by_index(begin = x1_47_begin_0, end = x1_47_end_0, end_mask = x1_47_end_mask_0, x = k_23)[name = string("x1_47")]; + tensor x2_47_begin_0 = const()[name = string("x2_47_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_47_end_0 = const()[name = string("x2_47_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_47_end_mask_0 = const()[name = string("x2_47_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_47 = slice_by_index(begin = x2_47_begin_0, end = x2_47_end_0, end_mask = x2_47_end_mask_0, x = k_23)[name = string("x2_47")]; + fp16 const_393_promoted = const()[name = string("const_393_promoted"), val = fp16(-0x1p+0)]; + tensor var_6956 = mul(x = x2_47, y = const_393_promoted)[name = string("op_6956")]; + int32 var_6958 = const()[name = string("op_6958"), val = int32(-1)]; + bool var_6959_interleave_0 = const()[name = string("op_6959_interleave_0"), val = bool(false)]; + tensor var_6959 = concat(axis = var_6958, interleave = var_6959_interleave_0, values = (var_6956, x1_47))[name = string("op_6959")]; + tensor var_6960 = mul(x = var_6959, y = sin_5)[name = string("op_6960")]; + tensor key_states_113 = add(x = var_6935, y = var_6960)[name = string("key_states_113")]; + tensor expand_dims_132 = const()[name = string("expand_dims_132"), val = tensor([25])]; + tensor expand_dims_133 = const()[name = string("expand_dims_133"), val = tensor([0])]; + tensor expand_dims_135 = const()[name = string("expand_dims_135"), val = tensor([0])]; + tensor expand_dims_136 = const()[name = string("expand_dims_136"), val = tensor([26])]; + int32 concat_200_axis_0 = const()[name = string("concat_200_axis_0"), val = int32(0)]; + bool concat_200_interleave_0 = const()[name = string("concat_200_interleave_0"), val = bool(false)]; + tensor concat_200 = concat(axis = concat_200_axis_0, interleave = concat_200_interleave_0, values = (expand_dims_132, expand_dims_133, current_pos, expand_dims_135))[name = string("concat_200")]; + tensor concat_201_values1_0 = const()[name = string("concat_201_values1_0"), val = tensor([0])]; + tensor concat_201_values3_0 = const()[name = string("concat_201_values3_0"), val = tensor([0])]; + int32 concat_201_axis_0 = const()[name = string("concat_201_axis_0"), val = int32(0)]; + bool concat_201_interleave_0 = const()[name = string("concat_201_interleave_0"), val = bool(false)]; + tensor concat_201 = concat(axis = concat_201_axis_0, interleave = concat_201_interleave_0, values = (expand_dims_136, concat_201_values1_0, var_1042, concat_201_values3_0))[name = string("concat_201")]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_200, begin_mask = model_model_kv_cache_0_internal_tensor_assign_23_begin_mask_0, end = concat_201, end_mask = model_model_kv_cache_0_internal_tensor_assign_23_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_23_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_23_stride_0, update = key_states_113, x = coreml_update_state_49)[name = string("model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_23_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_106_write_state")]; + tensor coreml_update_state_50 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_106")]; + tensor expand_dims_138 = const()[name = string("expand_dims_138"), val = tensor([53])]; + tensor expand_dims_139 = const()[name = string("expand_dims_139"), val = tensor([0])]; + tensor expand_dims_141 = const()[name = string("expand_dims_141"), val = tensor([0])]; + tensor expand_dims_142 = const()[name = string("expand_dims_142"), val = tensor([54])]; + int32 concat_204_axis_0 = const()[name = string("concat_204_axis_0"), val = int32(0)]; + bool concat_204_interleave_0 = const()[name = string("concat_204_interleave_0"), val = bool(false)]; + tensor concat_204 = concat(axis = concat_204_axis_0, interleave = concat_204_interleave_0, values = (expand_dims_138, expand_dims_139, current_pos, expand_dims_141))[name = string("concat_204")]; + tensor concat_205_values1_0 = const()[name = string("concat_205_values1_0"), val = tensor([0])]; + tensor concat_205_values3_0 = const()[name = string("concat_205_values3_0"), val = tensor([0])]; + int32 concat_205_axis_0 = const()[name = string("concat_205_axis_0"), val = int32(0)]; + bool concat_205_interleave_0 = const()[name = string("concat_205_interleave_0"), val = bool(false)]; + tensor concat_205 = concat(axis = concat_205_axis_0, interleave = concat_205_interleave_0, values = (expand_dims_142, concat_205_values1_0, var_1042, concat_205_values3_0))[name = string("concat_205")]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_91 = transpose(perm = var_6843, x = var_6838)[name = string("transpose_23")]; + tensor model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_204, begin_mask = model_model_kv_cache_0_internal_tensor_assign_24_begin_mask_0, end = concat_205, end_mask = model_model_kv_cache_0_internal_tensor_assign_24_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_24_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_24_stride_0, update = value_states_91, x = coreml_update_state_50)[name = string("model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_24_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_107_write_state")]; + tensor coreml_update_state_51 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_107")]; + tensor var_7031_begin_0 = const()[name = string("op_7031_begin_0"), val = tensor([25, 0, 0, 0])]; + tensor var_7031_end_0 = const()[name = string("op_7031_end_0"), val = tensor([26, 8, 1024, 128])]; + tensor var_7031_end_mask_0 = const()[name = string("op_7031_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_7031_cast_fp16 = slice_by_index(begin = var_7031_begin_0, end = var_7031_end_0, end_mask = var_7031_end_mask_0, x = coreml_update_state_51)[name = string("op_7031_cast_fp16")]; + tensor K_layer_cache_23_axes_0 = const()[name = string("K_layer_cache_23_axes_0"), val = tensor([0])]; + tensor K_layer_cache_23_cast_fp16 = squeeze(axes = K_layer_cache_23_axes_0, x = var_7031_cast_fp16)[name = string("K_layer_cache_23_cast_fp16")]; + tensor var_7038_begin_0 = const()[name = string("op_7038_begin_0"), val = tensor([53, 0, 0, 0])]; + tensor var_7038_end_0 = const()[name = string("op_7038_end_0"), val = tensor([54, 8, 1024, 128])]; + tensor var_7038_end_mask_0 = const()[name = string("op_7038_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_7038_cast_fp16 = slice_by_index(begin = var_7038_begin_0, end = var_7038_end_0, end_mask = var_7038_end_mask_0, x = coreml_update_state_51)[name = string("op_7038_cast_fp16")]; + tensor V_layer_cache_23_axes_0 = const()[name = string("V_layer_cache_23_axes_0"), val = tensor([0])]; + tensor V_layer_cache_23_cast_fp16 = squeeze(axes = V_layer_cache_23_axes_0, x = var_7038_cast_fp16)[name = string("V_layer_cache_23_cast_fp16")]; + tensor x_179_axes_0 = const()[name = string("x_179_axes_0"), val = tensor([1])]; + tensor x_179_cast_fp16 = expand_dims(axes = x_179_axes_0, x = K_layer_cache_23_cast_fp16)[name = string("x_179_cast_fp16")]; + tensor var_7067 = const()[name = string("op_7067"), val = tensor([1, 2, 1, 1])]; + tensor x_181_cast_fp16 = tile(reps = var_7067, x = x_179_cast_fp16)[name = string("x_181_cast_fp16")]; + tensor var_7079 = const()[name = string("op_7079"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_117_cast_fp16 = reshape(shape = var_7079, x = x_181_cast_fp16)[name = string("key_states_117_cast_fp16")]; + tensor x_185_axes_0 = const()[name = string("x_185_axes_0"), val = tensor([1])]; + tensor x_185_cast_fp16 = expand_dims(axes = x_185_axes_0, x = V_layer_cache_23_cast_fp16)[name = string("x_185_cast_fp16")]; + tensor var_7087 = const()[name = string("op_7087"), val = tensor([1, 2, 1, 1])]; + tensor x_187_cast_fp16 = tile(reps = var_7087, x = x_185_cast_fp16)[name = string("x_187_cast_fp16")]; + bool var_7114_transpose_x_0 = const()[name = string("op_7114_transpose_x_0"), val = bool(false)]; + bool var_7114_transpose_y_0 = const()[name = string("op_7114_transpose_y_0"), val = bool(true)]; + tensor var_7114 = matmul(transpose_x = var_7114_transpose_x_0, transpose_y = var_7114_transpose_y_0, x = query_states_91, y = key_states_117_cast_fp16)[name = string("op_7114")]; + fp16 var_7115_to_fp16 = const()[name = string("op_7115_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_45_cast_fp16 = mul(x = var_7114, y = var_7115_to_fp16)[name = string("attn_weights_45_cast_fp16")]; + tensor attn_weights_47_cast_fp16 = add(x = attn_weights_45_cast_fp16, y = causal_mask)[name = string("attn_weights_47_cast_fp16")]; + int32 var_7150 = const()[name = string("op_7150"), val = int32(-1)]; + tensor var_7152_cast_fp16 = softmax(axis = var_7150, x = attn_weights_47_cast_fp16)[name = string("op_7152_cast_fp16")]; + tensor concat_210 = const()[name = string("concat_210"), val = tensor([16, 128, 1024])]; + tensor reshape_33_cast_fp16 = reshape(shape = concat_210, x = var_7152_cast_fp16)[name = string("reshape_33_cast_fp16")]; + tensor concat_211 = const()[name = string("concat_211"), val = tensor([16, 1024, 128])]; + tensor reshape_34_cast_fp16 = reshape(shape = concat_211, x = x_187_cast_fp16)[name = string("reshape_34_cast_fp16")]; + bool matmul_11_transpose_x_0 = const()[name = string("matmul_11_transpose_x_0"), val = bool(false)]; + bool matmul_11_transpose_y_0 = const()[name = string("matmul_11_transpose_y_0"), val = bool(false)]; + tensor matmul_11_cast_fp16 = matmul(transpose_x = matmul_11_transpose_x_0, transpose_y = matmul_11_transpose_y_0, x = reshape_33_cast_fp16, y = reshape_34_cast_fp16)[name = string("matmul_11_cast_fp16")]; + tensor concat_215 = const()[name = string("concat_215"), val = tensor([1, 16, 128, 128])]; + tensor reshape_35_cast_fp16 = reshape(shape = concat_215, x = matmul_11_cast_fp16)[name = string("reshape_35_cast_fp16")]; + tensor var_7164_perm_0 = const()[name = string("op_7164_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_7183 = const()[name = string("op_7183"), val = tensor([1, 128, 2048])]; + tensor var_7164_cast_fp16 = transpose(perm = var_7164_perm_0, x = reshape_35_cast_fp16)[name = string("transpose_22")]; + tensor attn_output_115_cast_fp16 = reshape(shape = var_7183, x = var_7164_cast_fp16)[name = string("attn_output_115_cast_fp16")]; + tensor var_7188 = const()[name = string("op_7188"), val = tensor([0, 2, 1])]; + string var_7204_pad_type_0 = const()[name = string("op_7204_pad_type_0"), val = string("valid")]; + int32 var_7204_groups_0 = const()[name = string("op_7204_groups_0"), val = int32(1)]; + tensor var_7204_strides_0 = const()[name = string("op_7204_strides_0"), val = tensor([1])]; + tensor var_7204_pad_0 = const()[name = string("op_7204_pad_0"), val = tensor([0, 0])]; + tensor var_7204_dilations_0 = const()[name = string("op_7204_dilations_0"), val = tensor([1])]; + tensor squeeze_11_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(711181312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715375680))))[name = string("squeeze_11_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_7189_cast_fp16 = transpose(perm = var_7188, x = attn_output_115_cast_fp16)[name = string("transpose_21")]; + tensor var_7204_cast_fp16 = conv(dilations = var_7204_dilations_0, groups = var_7204_groups_0, pad = var_7204_pad_0, pad_type = var_7204_pad_type_0, strides = var_7204_strides_0, weight = squeeze_11_cast_fp16_to_fp32_to_fp16_palettized, x = var_7189_cast_fp16)[name = string("op_7204_cast_fp16")]; + tensor var_7208 = const()[name = string("op_7208"), val = tensor([0, 2, 1])]; + tensor attn_output_119_cast_fp16 = transpose(perm = var_7208, x = var_7204_cast_fp16)[name = string("transpose_20")]; + tensor hidden_states_119_cast_fp16 = add(x = hidden_states_111_cast_fp16, y = attn_output_119_cast_fp16)[name = string("hidden_states_119_cast_fp16")]; + int32 var_7221 = const()[name = string("op_7221"), val = int32(-1)]; + fp16 const_405_promoted_to_fp16 = const()[name = string("const_405_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7223_cast_fp16 = mul(x = hidden_states_119_cast_fp16, y = const_405_promoted_to_fp16)[name = string("op_7223_cast_fp16")]; + bool input_209_interleave_0 = const()[name = string("input_209_interleave_0"), val = bool(false)]; + tensor input_209_cast_fp16 = concat(axis = var_7221, interleave = input_209_interleave_0, values = (hidden_states_119_cast_fp16, var_7223_cast_fp16))[name = string("input_209_cast_fp16")]; + tensor normed_189_axes_0 = const()[name = string("normed_189_axes_0"), val = tensor([-1])]; + fp16 var_7218_to_fp16 = const()[name = string("op_7218_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_189_cast_fp16 = layer_norm(axes = normed_189_axes_0, epsilon = var_7218_to_fp16, x = input_209_cast_fp16)[name = string("normed_189_cast_fp16")]; + tensor normed_191_begin_0 = const()[name = string("normed_191_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_191_end_0 = const()[name = string("normed_191_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_191_end_mask_0 = const()[name = string("normed_191_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_191_cast_fp16 = slice_by_index(begin = normed_191_begin_0, end = normed_191_end_0, end_mask = normed_191_end_mask_0, x = normed_189_cast_fp16)[name = string("normed_191_cast_fp16")]; + tensor const_408_promoted_to_fp16 = const()[name = string("const_408_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715506816)))]; + tensor x_189_cast_fp16 = mul(x = normed_191_cast_fp16, y = const_408_promoted_to_fp16)[name = string("x_189_cast_fp16")]; + tensor var_7248 = const()[name = string("op_7248"), val = tensor([0, 2, 1])]; + tensor input_211_axes_0 = const()[name = string("input_211_axes_0"), val = tensor([2])]; + tensor var_7249 = transpose(perm = var_7248, x = x_189_cast_fp16)[name = string("transpose_19")]; + tensor input_211 = expand_dims(axes = input_211_axes_0, x = var_7249)[name = string("input_211")]; + string input_213_pad_type_0 = const()[name = string("input_213_pad_type_0"), val = string("valid")]; + tensor input_213_strides_0 = const()[name = string("input_213_strides_0"), val = tensor([1, 1])]; + tensor input_213_pad_0 = const()[name = string("input_213_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_213_dilations_0 = const()[name = string("input_213_dilations_0"), val = tensor([1, 1])]; + int32 input_213_groups_0 = const()[name = string("input_213_groups_0"), val = int32(1)]; + tensor input_213 = conv(dilations = input_213_dilations_0, groups = input_213_groups_0, pad = input_213_pad_0, pad_type = input_213_pad_type_0, strides = input_213_strides_0, weight = model_model_layers_25_mlp_gate_proj_weight_palettized, x = input_211)[name = string("input_213")]; + string b_23_pad_type_0 = const()[name = string("b_23_pad_type_0"), val = string("valid")]; + tensor b_23_strides_0 = const()[name = string("b_23_strides_0"), val = tensor([1, 1])]; + tensor b_23_pad_0 = const()[name = string("b_23_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_23_dilations_0 = const()[name = string("b_23_dilations_0"), val = tensor([1, 1])]; + int32 b_23_groups_0 = const()[name = string("b_23_groups_0"), val = int32(1)]; + tensor b_23 = conv(dilations = b_23_dilations_0, groups = b_23_groups_0, pad = b_23_pad_0, pad_type = b_23_pad_type_0, strides = b_23_strides_0, weight = model_model_layers_25_mlp_up_proj_weight_palettized, x = input_211)[name = string("b_23")]; + tensor c_23 = silu(x = input_213)[name = string("c_23")]; + tensor input_215 = mul(x = c_23, y = b_23)[name = string("input_215")]; + string e_23_pad_type_0 = const()[name = string("e_23_pad_type_0"), val = string("valid")]; + tensor e_23_strides_0 = const()[name = string("e_23_strides_0"), val = tensor([1, 1])]; + tensor e_23_pad_0 = const()[name = string("e_23_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_23_dilations_0 = const()[name = string("e_23_dilations_0"), val = tensor([1, 1])]; + int32 e_23_groups_0 = const()[name = string("e_23_groups_0"), val = int32(1)]; + tensor e_23 = conv(dilations = e_23_dilations_0, groups = e_23_groups_0, pad = e_23_pad_0, pad_type = e_23_pad_type_0, strides = e_23_strides_0, weight = model_model_layers_25_mlp_down_proj_weight_palettized, x = input_215)[name = string("e_23")]; + tensor var_7271_axes_0 = const()[name = string("op_7271_axes_0"), val = tensor([2])]; + tensor var_7271 = squeeze(axes = var_7271_axes_0, x = e_23)[name = string("op_7271")]; + tensor var_7272 = const()[name = string("op_7272"), val = tensor([0, 2, 1])]; + tensor var_7273 = transpose(perm = var_7272, x = var_7271)[name = string("transpose_18")]; + tensor hidden_states_121_cast_fp16 = add(x = hidden_states_119_cast_fp16, y = var_7273)[name = string("hidden_states_121_cast_fp16")]; + int32 var_7285 = const()[name = string("op_7285"), val = int32(-1)]; + fp16 const_409_promoted_to_fp16 = const()[name = string("const_409_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7287_cast_fp16 = mul(x = hidden_states_121_cast_fp16, y = const_409_promoted_to_fp16)[name = string("op_7287_cast_fp16")]; + bool input_217_interleave_0 = const()[name = string("input_217_interleave_0"), val = bool(false)]; + tensor input_217_cast_fp16 = concat(axis = var_7285, interleave = input_217_interleave_0, values = (hidden_states_121_cast_fp16, var_7287_cast_fp16))[name = string("input_217_cast_fp16")]; + tensor normed_193_axes_0 = const()[name = string("normed_193_axes_0"), val = tensor([-1])]; + fp16 var_7282_to_fp16 = const()[name = string("op_7282_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_193_cast_fp16 = layer_norm(axes = normed_193_axes_0, epsilon = var_7282_to_fp16, x = input_217_cast_fp16)[name = string("normed_193_cast_fp16")]; + tensor normed_195_begin_0 = const()[name = string("normed_195_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_195_end_0 = const()[name = string("normed_195_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_195_end_mask_0 = const()[name = string("normed_195_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_195_cast_fp16 = slice_by_index(begin = normed_195_begin_0, end = normed_195_end_0, end_mask = normed_195_end_mask_0, x = normed_193_cast_fp16)[name = string("normed_195_cast_fp16")]; + tensor const_412_promoted_to_fp16 = const()[name = string("const_412_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715510976)))]; + tensor hidden_states_123_cast_fp16 = mul(x = normed_195_cast_fp16, y = const_412_promoted_to_fp16)[name = string("hidden_states_123_cast_fp16")]; + tensor var_7310 = const()[name = string("op_7310"), val = tensor([0, 2, 1])]; + tensor var_7313_axes_0 = const()[name = string("op_7313_axes_0"), val = tensor([2])]; + tensor var_7311_cast_fp16 = transpose(perm = var_7310, x = hidden_states_123_cast_fp16)[name = string("transpose_17")]; + tensor var_7313_cast_fp16 = expand_dims(axes = var_7313_axes_0, x = var_7311_cast_fp16)[name = string("op_7313_cast_fp16")]; + string query_states_97_pad_type_0 = const()[name = string("query_states_97_pad_type_0"), val = string("valid")]; + tensor query_states_97_strides_0 = const()[name = string("query_states_97_strides_0"), val = tensor([1, 1])]; + tensor query_states_97_pad_0 = const()[name = string("query_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_97_dilations_0 = const()[name = string("query_states_97_dilations_0"), val = tensor([1, 1])]; + int32 query_states_97_groups_0 = const()[name = string("query_states_97_groups_0"), val = int32(1)]; + tensor query_states_97 = conv(dilations = query_states_97_dilations_0, groups = query_states_97_groups_0, pad = query_states_97_pad_0, pad_type = query_states_97_pad_type_0, strides = query_states_97_strides_0, weight = model_model_layers_26_self_attn_q_proj_weight_palettized, x = var_7313_cast_fp16)[name = string("query_states_97")]; + string key_states_121_pad_type_0 = const()[name = string("key_states_121_pad_type_0"), val = string("valid")]; + tensor key_states_121_strides_0 = const()[name = string("key_states_121_strides_0"), val = tensor([1, 1])]; + tensor key_states_121_pad_0 = const()[name = string("key_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_121_dilations_0 = const()[name = string("key_states_121_dilations_0"), val = tensor([1, 1])]; + int32 key_states_121_groups_0 = const()[name = string("key_states_121_groups_0"), val = int32(1)]; + tensor key_states_121 = conv(dilations = key_states_121_dilations_0, groups = key_states_121_groups_0, pad = key_states_121_pad_0, pad_type = key_states_121_pad_type_0, strides = key_states_121_strides_0, weight = model_model_layers_26_self_attn_k_proj_weight_palettized, x = var_7313_cast_fp16)[name = string("key_states_121")]; + string value_states_97_pad_type_0 = const()[name = string("value_states_97_pad_type_0"), val = string("valid")]; + tensor value_states_97_strides_0 = const()[name = string("value_states_97_strides_0"), val = tensor([1, 1])]; + tensor value_states_97_pad_0 = const()[name = string("value_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_97_dilations_0 = const()[name = string("value_states_97_dilations_0"), val = tensor([1, 1])]; + int32 value_states_97_groups_0 = const()[name = string("value_states_97_groups_0"), val = int32(1)]; + tensor value_states_97 = conv(dilations = value_states_97_dilations_0, groups = value_states_97_groups_0, pad = value_states_97_pad_0, pad_type = value_states_97_pad_type_0, strides = value_states_97_strides_0, weight = model_model_layers_26_self_attn_v_proj_weight_palettized, x = var_7313_cast_fp16)[name = string("value_states_97")]; + tensor var_7355 = const()[name = string("op_7355"), val = tensor([1, 16, 128, 128])]; + tensor var_7356 = reshape(shape = var_7355, x = query_states_97)[name = string("op_7356")]; + tensor var_7361 = const()[name = string("op_7361"), val = tensor([0, 1, 3, 2])]; + tensor var_7366 = const()[name = string("op_7366"), val = tensor([1, 8, 128, 128])]; + tensor var_7367 = reshape(shape = var_7366, x = key_states_121)[name = string("op_7367")]; + tensor var_7372 = const()[name = string("op_7372"), val = tensor([0, 1, 3, 2])]; + tensor var_7377 = const()[name = string("op_7377"), val = tensor([1, 8, 128, 128])]; + tensor var_7378 = reshape(shape = var_7377, x = value_states_97)[name = string("op_7378")]; + tensor var_7383 = const()[name = string("op_7383"), val = tensor([0, 1, 3, 2])]; + int32 var_7394 = const()[name = string("op_7394"), val = int32(-1)]; + fp16 const_414_promoted = const()[name = string("const_414_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_125 = transpose(perm = var_7361, x = var_7356)[name = string("transpose_16")]; + tensor var_7396 = mul(x = hidden_states_125, y = const_414_promoted)[name = string("op_7396")]; + bool input_221_interleave_0 = const()[name = string("input_221_interleave_0"), val = bool(false)]; + tensor input_221 = concat(axis = var_7394, interleave = input_221_interleave_0, values = (hidden_states_125, var_7396))[name = string("input_221")]; + tensor normed_197_axes_0 = const()[name = string("normed_197_axes_0"), val = tensor([-1])]; + fp16 var_7391_to_fp16 = const()[name = string("op_7391_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_197_cast_fp16 = layer_norm(axes = normed_197_axes_0, epsilon = var_7391_to_fp16, x = input_221)[name = string("normed_197_cast_fp16")]; + tensor normed_199_begin_0 = const()[name = string("normed_199_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_199_end_0 = const()[name = string("normed_199_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_199_end_mask_0 = const()[name = string("normed_199_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_199 = slice_by_index(begin = normed_199_begin_0, end = normed_199_end_0, end_mask = normed_199_end_mask_0, x = normed_197_cast_fp16)[name = string("normed_199")]; + tensor const_417 = const()[name = string("const_417"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715515136)))]; + tensor q_25 = mul(x = normed_199, y = const_417)[name = string("q_25")]; + int32 var_7419 = const()[name = string("op_7419"), val = int32(-1)]; + fp16 const_418_promoted = const()[name = string("const_418_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_127 = transpose(perm = var_7372, x = var_7367)[name = string("transpose_15")]; + tensor var_7421 = mul(x = hidden_states_127, y = const_418_promoted)[name = string("op_7421")]; + bool input_223_interleave_0 = const()[name = string("input_223_interleave_0"), val = bool(false)]; + tensor input_223 = concat(axis = var_7419, interleave = input_223_interleave_0, values = (hidden_states_127, var_7421))[name = string("input_223")]; + tensor normed_201_axes_0 = const()[name = string("normed_201_axes_0"), val = tensor([-1])]; + fp16 var_7416_to_fp16 = const()[name = string("op_7416_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_201_cast_fp16 = layer_norm(axes = normed_201_axes_0, epsilon = var_7416_to_fp16, x = input_223)[name = string("normed_201_cast_fp16")]; + tensor normed_203_begin_0 = const()[name = string("normed_203_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_203_end_0 = const()[name = string("normed_203_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_203_end_mask_0 = const()[name = string("normed_203_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_203 = slice_by_index(begin = normed_203_begin_0, end = normed_203_end_0, end_mask = normed_203_end_mask_0, x = normed_201_cast_fp16)[name = string("normed_203")]; + tensor const_421 = const()[name = string("const_421"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715515456)))]; + tensor k_25 = mul(x = normed_203, y = const_421)[name = string("k_25")]; + tensor var_7447 = mul(x = q_25, y = cos_5)[name = string("op_7447")]; + tensor x1_49_begin_0 = const()[name = string("x1_49_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_49_end_0 = const()[name = string("x1_49_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_49_end_mask_0 = const()[name = string("x1_49_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_49 = slice_by_index(begin = x1_49_begin_0, end = x1_49_end_0, end_mask = x1_49_end_mask_0, x = q_25)[name = string("x1_49")]; + tensor x2_49_begin_0 = const()[name = string("x2_49_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_49_end_0 = const()[name = string("x2_49_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_49_end_mask_0 = const()[name = string("x2_49_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_49 = slice_by_index(begin = x2_49_begin_0, end = x2_49_end_0, end_mask = x2_49_end_mask_0, x = q_25)[name = string("x2_49")]; + fp16 const_424_promoted = const()[name = string("const_424_promoted"), val = fp16(-0x1p+0)]; + tensor var_7468 = mul(x = x2_49, y = const_424_promoted)[name = string("op_7468")]; + int32 var_7470 = const()[name = string("op_7470"), val = int32(-1)]; + bool var_7471_interleave_0 = const()[name = string("op_7471_interleave_0"), val = bool(false)]; + tensor var_7471 = concat(axis = var_7470, interleave = var_7471_interleave_0, values = (var_7468, x1_49))[name = string("op_7471")]; + tensor var_7472 = mul(x = var_7471, y = sin_5)[name = string("op_7472")]; + tensor query_states_99 = add(x = var_7447, y = var_7472)[name = string("query_states_99")]; + tensor var_7475 = mul(x = k_25, y = cos_5)[name = string("op_7475")]; + tensor x1_51_begin_0 = const()[name = string("x1_51_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_51_end_0 = const()[name = string("x1_51_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_51_end_mask_0 = const()[name = string("x1_51_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_51 = slice_by_index(begin = x1_51_begin_0, end = x1_51_end_0, end_mask = x1_51_end_mask_0, x = k_25)[name = string("x1_51")]; + tensor x2_51_begin_0 = const()[name = string("x2_51_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_51_end_0 = const()[name = string("x2_51_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_51_end_mask_0 = const()[name = string("x2_51_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_51 = slice_by_index(begin = x2_51_begin_0, end = x2_51_end_0, end_mask = x2_51_end_mask_0, x = k_25)[name = string("x2_51")]; + fp16 const_427_promoted = const()[name = string("const_427_promoted"), val = fp16(-0x1p+0)]; + tensor var_7496 = mul(x = x2_51, y = const_427_promoted)[name = string("op_7496")]; + int32 var_7498 = const()[name = string("op_7498"), val = int32(-1)]; + bool var_7499_interleave_0 = const()[name = string("op_7499_interleave_0"), val = bool(false)]; + tensor var_7499 = concat(axis = var_7498, interleave = var_7499_interleave_0, values = (var_7496, x1_51))[name = string("op_7499")]; + tensor var_7500 = mul(x = var_7499, y = sin_5)[name = string("op_7500")]; + tensor key_states_123 = add(x = var_7475, y = var_7500)[name = string("key_states_123")]; + tensor expand_dims_144 = const()[name = string("expand_dims_144"), val = tensor([26])]; + tensor expand_dims_145 = const()[name = string("expand_dims_145"), val = tensor([0])]; + tensor expand_dims_147 = const()[name = string("expand_dims_147"), val = tensor([0])]; + tensor expand_dims_148 = const()[name = string("expand_dims_148"), val = tensor([27])]; + int32 concat_218_axis_0 = const()[name = string("concat_218_axis_0"), val = int32(0)]; + bool concat_218_interleave_0 = const()[name = string("concat_218_interleave_0"), val = bool(false)]; + tensor concat_218 = concat(axis = concat_218_axis_0, interleave = concat_218_interleave_0, values = (expand_dims_144, expand_dims_145, current_pos, expand_dims_147))[name = string("concat_218")]; + tensor concat_219_values1_0 = const()[name = string("concat_219_values1_0"), val = tensor([0])]; + tensor concat_219_values3_0 = const()[name = string("concat_219_values3_0"), val = tensor([0])]; + int32 concat_219_axis_0 = const()[name = string("concat_219_axis_0"), val = int32(0)]; + bool concat_219_interleave_0 = const()[name = string("concat_219_interleave_0"), val = bool(false)]; + tensor concat_219 = concat(axis = concat_219_axis_0, interleave = concat_219_interleave_0, values = (expand_dims_148, concat_219_values1_0, var_1042, concat_219_values3_0))[name = string("concat_219")]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_218, begin_mask = model_model_kv_cache_0_internal_tensor_assign_25_begin_mask_0, end = concat_219, end_mask = model_model_kv_cache_0_internal_tensor_assign_25_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_25_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_25_stride_0, update = key_states_123, x = coreml_update_state_51)[name = string("model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_25_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_108_write_state")]; + tensor coreml_update_state_52 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_108")]; + tensor expand_dims_150 = const()[name = string("expand_dims_150"), val = tensor([54])]; + tensor expand_dims_151 = const()[name = string("expand_dims_151"), val = tensor([0])]; + tensor expand_dims_153 = const()[name = string("expand_dims_153"), val = tensor([0])]; + tensor expand_dims_154 = const()[name = string("expand_dims_154"), val = tensor([55])]; + int32 concat_222_axis_0 = const()[name = string("concat_222_axis_0"), val = int32(0)]; + bool concat_222_interleave_0 = const()[name = string("concat_222_interleave_0"), val = bool(false)]; + tensor concat_222 = concat(axis = concat_222_axis_0, interleave = concat_222_interleave_0, values = (expand_dims_150, expand_dims_151, current_pos, expand_dims_153))[name = string("concat_222")]; + tensor concat_223_values1_0 = const()[name = string("concat_223_values1_0"), val = tensor([0])]; + tensor concat_223_values3_0 = const()[name = string("concat_223_values3_0"), val = tensor([0])]; + int32 concat_223_axis_0 = const()[name = string("concat_223_axis_0"), val = int32(0)]; + bool concat_223_interleave_0 = const()[name = string("concat_223_interleave_0"), val = bool(false)]; + tensor concat_223 = concat(axis = concat_223_axis_0, interleave = concat_223_interleave_0, values = (expand_dims_154, concat_223_values1_0, var_1042, concat_223_values3_0))[name = string("concat_223")]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_99 = transpose(perm = var_7383, x = var_7378)[name = string("transpose_14")]; + tensor model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_222, begin_mask = model_model_kv_cache_0_internal_tensor_assign_26_begin_mask_0, end = concat_223, end_mask = model_model_kv_cache_0_internal_tensor_assign_26_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_26_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_26_stride_0, update = value_states_99, x = coreml_update_state_52)[name = string("model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_26_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_109_write_state")]; + tensor coreml_update_state_53 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_109")]; + tensor var_7571_begin_0 = const()[name = string("op_7571_begin_0"), val = tensor([26, 0, 0, 0])]; + tensor var_7571_end_0 = const()[name = string("op_7571_end_0"), val = tensor([27, 8, 1024, 128])]; + tensor var_7571_end_mask_0 = const()[name = string("op_7571_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_7571_cast_fp16 = slice_by_index(begin = var_7571_begin_0, end = var_7571_end_0, end_mask = var_7571_end_mask_0, x = coreml_update_state_53)[name = string("op_7571_cast_fp16")]; + tensor K_layer_cache_25_axes_0 = const()[name = string("K_layer_cache_25_axes_0"), val = tensor([0])]; + tensor K_layer_cache_25_cast_fp16 = squeeze(axes = K_layer_cache_25_axes_0, x = var_7571_cast_fp16)[name = string("K_layer_cache_25_cast_fp16")]; + tensor var_7578_begin_0 = const()[name = string("op_7578_begin_0"), val = tensor([54, 0, 0, 0])]; + tensor var_7578_end_0 = const()[name = string("op_7578_end_0"), val = tensor([55, 8, 1024, 128])]; + tensor var_7578_end_mask_0 = const()[name = string("op_7578_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_7578_cast_fp16 = slice_by_index(begin = var_7578_begin_0, end = var_7578_end_0, end_mask = var_7578_end_mask_0, x = coreml_update_state_53)[name = string("op_7578_cast_fp16")]; + tensor V_layer_cache_25_axes_0 = const()[name = string("V_layer_cache_25_axes_0"), val = tensor([0])]; + tensor V_layer_cache_25_cast_fp16 = squeeze(axes = V_layer_cache_25_axes_0, x = var_7578_cast_fp16)[name = string("V_layer_cache_25_cast_fp16")]; + tensor x_195_axes_0 = const()[name = string("x_195_axes_0"), val = tensor([1])]; + tensor x_195_cast_fp16 = expand_dims(axes = x_195_axes_0, x = K_layer_cache_25_cast_fp16)[name = string("x_195_cast_fp16")]; + tensor var_7607 = const()[name = string("op_7607"), val = tensor([1, 2, 1, 1])]; + tensor x_197_cast_fp16 = tile(reps = var_7607, x = x_195_cast_fp16)[name = string("x_197_cast_fp16")]; + tensor var_7619 = const()[name = string("op_7619"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_127_cast_fp16 = reshape(shape = var_7619, x = x_197_cast_fp16)[name = string("key_states_127_cast_fp16")]; + tensor x_201_axes_0 = const()[name = string("x_201_axes_0"), val = tensor([1])]; + tensor x_201_cast_fp16 = expand_dims(axes = x_201_axes_0, x = V_layer_cache_25_cast_fp16)[name = string("x_201_cast_fp16")]; + tensor var_7627 = const()[name = string("op_7627"), val = tensor([1, 2, 1, 1])]; + tensor x_203_cast_fp16 = tile(reps = var_7627, x = x_201_cast_fp16)[name = string("x_203_cast_fp16")]; + bool var_7654_transpose_x_0 = const()[name = string("op_7654_transpose_x_0"), val = bool(false)]; + bool var_7654_transpose_y_0 = const()[name = string("op_7654_transpose_y_0"), val = bool(true)]; + tensor var_7654 = matmul(transpose_x = var_7654_transpose_x_0, transpose_y = var_7654_transpose_y_0, x = query_states_99, y = key_states_127_cast_fp16)[name = string("op_7654")]; + fp16 var_7655_to_fp16 = const()[name = string("op_7655_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_49_cast_fp16 = mul(x = var_7654, y = var_7655_to_fp16)[name = string("attn_weights_49_cast_fp16")]; + tensor attn_weights_51_cast_fp16 = add(x = attn_weights_49_cast_fp16, y = causal_mask)[name = string("attn_weights_51_cast_fp16")]; + int32 var_7690 = const()[name = string("op_7690"), val = int32(-1)]; + tensor var_7692_cast_fp16 = softmax(axis = var_7690, x = attn_weights_51_cast_fp16)[name = string("op_7692_cast_fp16")]; + tensor concat_228 = const()[name = string("concat_228"), val = tensor([16, 128, 1024])]; + tensor reshape_36_cast_fp16 = reshape(shape = concat_228, x = var_7692_cast_fp16)[name = string("reshape_36_cast_fp16")]; + tensor concat_229 = const()[name = string("concat_229"), val = tensor([16, 1024, 128])]; + tensor reshape_37_cast_fp16 = reshape(shape = concat_229, x = x_203_cast_fp16)[name = string("reshape_37_cast_fp16")]; + bool matmul_12_transpose_x_0 = const()[name = string("matmul_12_transpose_x_0"), val = bool(false)]; + bool matmul_12_transpose_y_0 = const()[name = string("matmul_12_transpose_y_0"), val = bool(false)]; + tensor matmul_12_cast_fp16 = matmul(transpose_x = matmul_12_transpose_x_0, transpose_y = matmul_12_transpose_y_0, x = reshape_36_cast_fp16, y = reshape_37_cast_fp16)[name = string("matmul_12_cast_fp16")]; + tensor concat_233 = const()[name = string("concat_233"), val = tensor([1, 16, 128, 128])]; + tensor reshape_38_cast_fp16 = reshape(shape = concat_233, x = matmul_12_cast_fp16)[name = string("reshape_38_cast_fp16")]; + tensor var_7704_perm_0 = const()[name = string("op_7704_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_7723 = const()[name = string("op_7723"), val = tensor([1, 128, 2048])]; + tensor var_7704_cast_fp16 = transpose(perm = var_7704_perm_0, x = reshape_38_cast_fp16)[name = string("transpose_13")]; + tensor attn_output_125_cast_fp16 = reshape(shape = var_7723, x = var_7704_cast_fp16)[name = string("attn_output_125_cast_fp16")]; + tensor var_7728 = const()[name = string("op_7728"), val = tensor([0, 2, 1])]; + string var_7744_pad_type_0 = const()[name = string("op_7744_pad_type_0"), val = string("valid")]; + int32 var_7744_groups_0 = const()[name = string("op_7744_groups_0"), val = int32(1)]; + tensor var_7744_strides_0 = const()[name = string("op_7744_strides_0"), val = tensor([1])]; + tensor var_7744_pad_0 = const()[name = string("op_7744_pad_0"), val = tensor([0, 0])]; + tensor var_7744_dilations_0 = const()[name = string("op_7744_dilations_0"), val = tensor([1])]; + tensor squeeze_12_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(715515776))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719710144))))[name = string("squeeze_12_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_7729_cast_fp16 = transpose(perm = var_7728, x = attn_output_125_cast_fp16)[name = string("transpose_12")]; + tensor var_7744_cast_fp16 = conv(dilations = var_7744_dilations_0, groups = var_7744_groups_0, pad = var_7744_pad_0, pad_type = var_7744_pad_type_0, strides = var_7744_strides_0, weight = squeeze_12_cast_fp16_to_fp32_to_fp16_palettized, x = var_7729_cast_fp16)[name = string("op_7744_cast_fp16")]; + tensor var_7748 = const()[name = string("op_7748"), val = tensor([0, 2, 1])]; + tensor attn_output_129_cast_fp16 = transpose(perm = var_7748, x = var_7744_cast_fp16)[name = string("transpose_11")]; + tensor hidden_states_129_cast_fp16 = add(x = hidden_states_121_cast_fp16, y = attn_output_129_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; + int32 var_7761 = const()[name = string("op_7761"), val = int32(-1)]; + fp16 const_439_promoted_to_fp16 = const()[name = string("const_439_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7763_cast_fp16 = mul(x = hidden_states_129_cast_fp16, y = const_439_promoted_to_fp16)[name = string("op_7763_cast_fp16")]; + bool input_227_interleave_0 = const()[name = string("input_227_interleave_0"), val = bool(false)]; + tensor input_227_cast_fp16 = concat(axis = var_7761, interleave = input_227_interleave_0, values = (hidden_states_129_cast_fp16, var_7763_cast_fp16))[name = string("input_227_cast_fp16")]; + tensor normed_205_axes_0 = const()[name = string("normed_205_axes_0"), val = tensor([-1])]; + fp16 var_7758_to_fp16 = const()[name = string("op_7758_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_205_cast_fp16 = layer_norm(axes = normed_205_axes_0, epsilon = var_7758_to_fp16, x = input_227_cast_fp16)[name = string("normed_205_cast_fp16")]; + tensor normed_207_begin_0 = const()[name = string("normed_207_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_207_end_0 = const()[name = string("normed_207_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_207_end_mask_0 = const()[name = string("normed_207_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_207_cast_fp16 = slice_by_index(begin = normed_207_begin_0, end = normed_207_end_0, end_mask = normed_207_end_mask_0, x = normed_205_cast_fp16)[name = string("normed_207_cast_fp16")]; + tensor const_442_promoted_to_fp16 = const()[name = string("const_442_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719841280)))]; + tensor x_205_cast_fp16 = mul(x = normed_207_cast_fp16, y = const_442_promoted_to_fp16)[name = string("x_205_cast_fp16")]; + tensor var_7788 = const()[name = string("op_7788"), val = tensor([0, 2, 1])]; + tensor input_229_axes_0 = const()[name = string("input_229_axes_0"), val = tensor([2])]; + tensor var_7789 = transpose(perm = var_7788, x = x_205_cast_fp16)[name = string("transpose_10")]; + tensor input_229 = expand_dims(axes = input_229_axes_0, x = var_7789)[name = string("input_229")]; + string input_231_pad_type_0 = const()[name = string("input_231_pad_type_0"), val = string("valid")]; + tensor input_231_strides_0 = const()[name = string("input_231_strides_0"), val = tensor([1, 1])]; + tensor input_231_pad_0 = const()[name = string("input_231_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_231_dilations_0 = const()[name = string("input_231_dilations_0"), val = tensor([1, 1])]; + int32 input_231_groups_0 = const()[name = string("input_231_groups_0"), val = int32(1)]; + tensor input_231 = conv(dilations = input_231_dilations_0, groups = input_231_groups_0, pad = input_231_pad_0, pad_type = input_231_pad_type_0, strides = input_231_strides_0, weight = model_model_layers_26_mlp_gate_proj_weight_palettized, x = input_229)[name = string("input_231")]; + string b_25_pad_type_0 = const()[name = string("b_25_pad_type_0"), val = string("valid")]; + tensor b_25_strides_0 = const()[name = string("b_25_strides_0"), val = tensor([1, 1])]; + tensor b_25_pad_0 = const()[name = string("b_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_25_dilations_0 = const()[name = string("b_25_dilations_0"), val = tensor([1, 1])]; + int32 b_25_groups_0 = const()[name = string("b_25_groups_0"), val = int32(1)]; + tensor b_25 = conv(dilations = b_25_dilations_0, groups = b_25_groups_0, pad = b_25_pad_0, pad_type = b_25_pad_type_0, strides = b_25_strides_0, weight = model_model_layers_26_mlp_up_proj_weight_palettized, x = input_229)[name = string("b_25")]; + tensor c_25 = silu(x = input_231)[name = string("c_25")]; + tensor input_233 = mul(x = c_25, y = b_25)[name = string("input_233")]; + string e_25_pad_type_0 = const()[name = string("e_25_pad_type_0"), val = string("valid")]; + tensor e_25_strides_0 = const()[name = string("e_25_strides_0"), val = tensor([1, 1])]; + tensor e_25_pad_0 = const()[name = string("e_25_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_25_dilations_0 = const()[name = string("e_25_dilations_0"), val = tensor([1, 1])]; + int32 e_25_groups_0 = const()[name = string("e_25_groups_0"), val = int32(1)]; + tensor e_25 = conv(dilations = e_25_dilations_0, groups = e_25_groups_0, pad = e_25_pad_0, pad_type = e_25_pad_type_0, strides = e_25_strides_0, weight = model_model_layers_26_mlp_down_proj_weight_palettized, x = input_233)[name = string("e_25")]; + tensor var_7811_axes_0 = const()[name = string("op_7811_axes_0"), val = tensor([2])]; + tensor var_7811 = squeeze(axes = var_7811_axes_0, x = e_25)[name = string("op_7811")]; + tensor var_7812 = const()[name = string("op_7812"), val = tensor([0, 2, 1])]; + tensor var_7813 = transpose(perm = var_7812, x = var_7811)[name = string("transpose_9")]; + tensor hidden_states_131_cast_fp16 = add(x = hidden_states_129_cast_fp16, y = var_7813)[name = string("hidden_states_131_cast_fp16")]; + int32 var_7825 = const()[name = string("op_7825"), val = int32(-1)]; + fp16 const_443_promoted_to_fp16 = const()[name = string("const_443_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_7827_cast_fp16 = mul(x = hidden_states_131_cast_fp16, y = const_443_promoted_to_fp16)[name = string("op_7827_cast_fp16")]; + bool input_235_interleave_0 = const()[name = string("input_235_interleave_0"), val = bool(false)]; + tensor input_235_cast_fp16 = concat(axis = var_7825, interleave = input_235_interleave_0, values = (hidden_states_131_cast_fp16, var_7827_cast_fp16))[name = string("input_235_cast_fp16")]; + tensor normed_209_axes_0 = const()[name = string("normed_209_axes_0"), val = tensor([-1])]; + fp16 var_7822_to_fp16 = const()[name = string("op_7822_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_209_cast_fp16 = layer_norm(axes = normed_209_axes_0, epsilon = var_7822_to_fp16, x = input_235_cast_fp16)[name = string("normed_209_cast_fp16")]; + tensor normed_211_begin_0 = const()[name = string("normed_211_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_211_end_0 = const()[name = string("normed_211_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_211_end_mask_0 = const()[name = string("normed_211_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_211_cast_fp16 = slice_by_index(begin = normed_211_begin_0, end = normed_211_end_0, end_mask = normed_211_end_mask_0, x = normed_209_cast_fp16)[name = string("normed_211_cast_fp16")]; + tensor const_446_promoted_to_fp16 = const()[name = string("const_446_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719845440)))]; + tensor hidden_states_133_cast_fp16 = mul(x = normed_211_cast_fp16, y = const_446_promoted_to_fp16)[name = string("hidden_states_133_cast_fp16")]; + tensor var_7850 = const()[name = string("op_7850"), val = tensor([0, 2, 1])]; + tensor var_7853_axes_0 = const()[name = string("op_7853_axes_0"), val = tensor([2])]; + tensor var_7851_cast_fp16 = transpose(perm = var_7850, x = hidden_states_133_cast_fp16)[name = string("transpose_8")]; + tensor var_7853_cast_fp16 = expand_dims(axes = var_7853_axes_0, x = var_7851_cast_fp16)[name = string("op_7853_cast_fp16")]; + string query_states_105_pad_type_0 = const()[name = string("query_states_105_pad_type_0"), val = string("valid")]; + tensor query_states_105_strides_0 = const()[name = string("query_states_105_strides_0"), val = tensor([1, 1])]; + tensor query_states_105_pad_0 = const()[name = string("query_states_105_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor query_states_105_dilations_0 = const()[name = string("query_states_105_dilations_0"), val = tensor([1, 1])]; + int32 query_states_105_groups_0 = const()[name = string("query_states_105_groups_0"), val = int32(1)]; + tensor query_states_105 = conv(dilations = query_states_105_dilations_0, groups = query_states_105_groups_0, pad = query_states_105_pad_0, pad_type = query_states_105_pad_type_0, strides = query_states_105_strides_0, weight = model_model_layers_27_self_attn_q_proj_weight_palettized, x = var_7853_cast_fp16)[name = string("query_states_105")]; + string key_states_131_pad_type_0 = const()[name = string("key_states_131_pad_type_0"), val = string("valid")]; + tensor key_states_131_strides_0 = const()[name = string("key_states_131_strides_0"), val = tensor([1, 1])]; + tensor key_states_131_pad_0 = const()[name = string("key_states_131_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor key_states_131_dilations_0 = const()[name = string("key_states_131_dilations_0"), val = tensor([1, 1])]; + int32 key_states_131_groups_0 = const()[name = string("key_states_131_groups_0"), val = int32(1)]; + tensor key_states_131 = conv(dilations = key_states_131_dilations_0, groups = key_states_131_groups_0, pad = key_states_131_pad_0, pad_type = key_states_131_pad_type_0, strides = key_states_131_strides_0, weight = model_model_layers_27_self_attn_k_proj_weight_palettized, x = var_7853_cast_fp16)[name = string("key_states_131")]; + string value_states_105_pad_type_0 = const()[name = string("value_states_105_pad_type_0"), val = string("valid")]; + tensor value_states_105_strides_0 = const()[name = string("value_states_105_strides_0"), val = tensor([1, 1])]; + tensor value_states_105_pad_0 = const()[name = string("value_states_105_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor value_states_105_dilations_0 = const()[name = string("value_states_105_dilations_0"), val = tensor([1, 1])]; + int32 value_states_105_groups_0 = const()[name = string("value_states_105_groups_0"), val = int32(1)]; + tensor value_states_105 = conv(dilations = value_states_105_dilations_0, groups = value_states_105_groups_0, pad = value_states_105_pad_0, pad_type = value_states_105_pad_type_0, strides = value_states_105_strides_0, weight = model_model_layers_27_self_attn_v_proj_weight_palettized, x = var_7853_cast_fp16)[name = string("value_states_105")]; + tensor var_7895 = const()[name = string("op_7895"), val = tensor([1, 16, 128, 128])]; + tensor var_7896 = reshape(shape = var_7895, x = query_states_105)[name = string("op_7896")]; + tensor var_7901 = const()[name = string("op_7901"), val = tensor([0, 1, 3, 2])]; + tensor var_7906 = const()[name = string("op_7906"), val = tensor([1, 8, 128, 128])]; + tensor var_7907 = reshape(shape = var_7906, x = key_states_131)[name = string("op_7907")]; + tensor var_7912 = const()[name = string("op_7912"), val = tensor([0, 1, 3, 2])]; + tensor var_7917 = const()[name = string("op_7917"), val = tensor([1, 8, 128, 128])]; + tensor var_7918 = reshape(shape = var_7917, x = value_states_105)[name = string("op_7918")]; + tensor var_7923 = const()[name = string("op_7923"), val = tensor([0, 1, 3, 2])]; + int32 var_7934 = const()[name = string("op_7934"), val = int32(-1)]; + fp16 const_448_promoted = const()[name = string("const_448_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_135 = transpose(perm = var_7901, x = var_7896)[name = string("transpose_7")]; + tensor var_7936 = mul(x = hidden_states_135, y = const_448_promoted)[name = string("op_7936")]; + bool input_239_interleave_0 = const()[name = string("input_239_interleave_0"), val = bool(false)]; + tensor input_239 = concat(axis = var_7934, interleave = input_239_interleave_0, values = (hidden_states_135, var_7936))[name = string("input_239")]; + tensor normed_213_axes_0 = const()[name = string("normed_213_axes_0"), val = tensor([-1])]; + fp16 var_7931_to_fp16 = const()[name = string("op_7931_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_213_cast_fp16 = layer_norm(axes = normed_213_axes_0, epsilon = var_7931_to_fp16, x = input_239)[name = string("normed_213_cast_fp16")]; + tensor normed_215_begin_0 = const()[name = string("normed_215_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_215_end_0 = const()[name = string("normed_215_end_0"), val = tensor([1, 16, 128, 128])]; + tensor normed_215_end_mask_0 = const()[name = string("normed_215_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_215 = slice_by_index(begin = normed_215_begin_0, end = normed_215_end_0, end_mask = normed_215_end_mask_0, x = normed_213_cast_fp16)[name = string("normed_215")]; + tensor const_451 = const()[name = string("const_451"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719849600)))]; + tensor q = mul(x = normed_215, y = const_451)[name = string("q")]; + int32 var_7959 = const()[name = string("op_7959"), val = int32(-1)]; + fp16 const_452_promoted = const()[name = string("const_452_promoted"), val = fp16(-0x1p+0)]; + tensor hidden_states_137 = transpose(perm = var_7912, x = var_7907)[name = string("transpose_6")]; + tensor var_7961 = mul(x = hidden_states_137, y = const_452_promoted)[name = string("op_7961")]; + bool input_241_interleave_0 = const()[name = string("input_241_interleave_0"), val = bool(false)]; + tensor input_241 = concat(axis = var_7959, interleave = input_241_interleave_0, values = (hidden_states_137, var_7961))[name = string("input_241")]; + tensor normed_217_axes_0 = const()[name = string("normed_217_axes_0"), val = tensor([-1])]; + fp16 var_7956_to_fp16 = const()[name = string("op_7956_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_217_cast_fp16 = layer_norm(axes = normed_217_axes_0, epsilon = var_7956_to_fp16, x = input_241)[name = string("normed_217_cast_fp16")]; + tensor normed_219_begin_0 = const()[name = string("normed_219_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor normed_219_end_0 = const()[name = string("normed_219_end_0"), val = tensor([1, 8, 128, 128])]; + tensor normed_219_end_mask_0 = const()[name = string("normed_219_end_mask_0"), val = tensor([true, true, true, false])]; + tensor normed_219 = slice_by_index(begin = normed_219_begin_0, end = normed_219_end_0, end_mask = normed_219_end_mask_0, x = normed_217_cast_fp16)[name = string("normed_219")]; + tensor const_455 = const()[name = string("const_455"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719849920)))]; + tensor k = mul(x = normed_219, y = const_455)[name = string("k")]; + tensor var_7987 = mul(x = q, y = cos_5)[name = string("op_7987")]; + tensor x1_53_begin_0 = const()[name = string("x1_53_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_53_end_0 = const()[name = string("x1_53_end_0"), val = tensor([1, 16, 128, 64])]; + tensor x1_53_end_mask_0 = const()[name = string("x1_53_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1_53 = slice_by_index(begin = x1_53_begin_0, end = x1_53_end_0, end_mask = x1_53_end_mask_0, x = q)[name = string("x1_53")]; + tensor x2_53_begin_0 = const()[name = string("x2_53_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_53_end_0 = const()[name = string("x2_53_end_0"), val = tensor([1, 16, 128, 128])]; + tensor x2_53_end_mask_0 = const()[name = string("x2_53_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2_53 = slice_by_index(begin = x2_53_begin_0, end = x2_53_end_0, end_mask = x2_53_end_mask_0, x = q)[name = string("x2_53")]; + fp16 const_458_promoted = const()[name = string("const_458_promoted"), val = fp16(-0x1p+0)]; + tensor var_8008 = mul(x = x2_53, y = const_458_promoted)[name = string("op_8008")]; + int32 var_8010 = const()[name = string("op_8010"), val = int32(-1)]; + bool var_8011_interleave_0 = const()[name = string("op_8011_interleave_0"), val = bool(false)]; + tensor var_8011 = concat(axis = var_8010, interleave = var_8011_interleave_0, values = (var_8008, x1_53))[name = string("op_8011")]; + tensor var_8012 = mul(x = var_8011, y = sin_5)[name = string("op_8012")]; + tensor query_states_107 = add(x = var_7987, y = var_8012)[name = string("query_states_107")]; + tensor var_8015 = mul(x = k, y = cos_5)[name = string("op_8015")]; + tensor x1_begin_0 = const()[name = string("x1_begin_0"), val = tensor([0, 0, 0, 0])]; + tensor x1_end_0 = const()[name = string("x1_end_0"), val = tensor([1, 8, 128, 64])]; + tensor x1_end_mask_0 = const()[name = string("x1_end_mask_0"), val = tensor([true, true, true, false])]; + tensor x1 = slice_by_index(begin = x1_begin_0, end = x1_end_0, end_mask = x1_end_mask_0, x = k)[name = string("x1")]; + tensor x2_begin_0 = const()[name = string("x2_begin_0"), val = tensor([0, 0, 0, 64])]; + tensor x2_end_0 = const()[name = string("x2_end_0"), val = tensor([1, 8, 128, 128])]; + tensor x2_end_mask_0 = const()[name = string("x2_end_mask_0"), val = tensor([true, true, true, true])]; + tensor x2 = slice_by_index(begin = x2_begin_0, end = x2_end_0, end_mask = x2_end_mask_0, x = k)[name = string("x2")]; + fp16 const_461_promoted = const()[name = string("const_461_promoted"), val = fp16(-0x1p+0)]; + tensor var_8036 = mul(x = x2, y = const_461_promoted)[name = string("op_8036")]; + int32 var_8038 = const()[name = string("op_8038"), val = int32(-1)]; + bool var_8039_interleave_0 = const()[name = string("op_8039_interleave_0"), val = bool(false)]; + tensor var_8039 = concat(axis = var_8038, interleave = var_8039_interleave_0, values = (var_8036, x1))[name = string("op_8039")]; + tensor var_8040 = mul(x = var_8039, y = sin_5)[name = string("op_8040")]; + tensor key_states_133 = add(x = var_8015, y = var_8040)[name = string("key_states_133")]; + tensor expand_dims_156 = const()[name = string("expand_dims_156"), val = tensor([27])]; + tensor expand_dims_157 = const()[name = string("expand_dims_157"), val = tensor([0])]; + tensor expand_dims_159 = const()[name = string("expand_dims_159"), val = tensor([0])]; + tensor expand_dims_160 = const()[name = string("expand_dims_160"), val = tensor([28])]; + int32 concat_236_axis_0 = const()[name = string("concat_236_axis_0"), val = int32(0)]; + bool concat_236_interleave_0 = const()[name = string("concat_236_interleave_0"), val = bool(false)]; + tensor concat_236 = concat(axis = concat_236_axis_0, interleave = concat_236_interleave_0, values = (expand_dims_156, expand_dims_157, current_pos, expand_dims_159))[name = string("concat_236")]; + tensor concat_237_values1_0 = const()[name = string("concat_237_values1_0"), val = tensor([0])]; + tensor concat_237_values3_0 = const()[name = string("concat_237_values3_0"), val = tensor([0])]; + int32 concat_237_axis_0 = const()[name = string("concat_237_axis_0"), val = int32(0)]; + bool concat_237_interleave_0 = const()[name = string("concat_237_interleave_0"), val = bool(false)]; + tensor concat_237 = concat(axis = concat_237_axis_0, interleave = concat_237_interleave_0, values = (expand_dims_160, concat_237_values1_0, var_1042, concat_237_values3_0))[name = string("concat_237")]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_236, begin_mask = model_model_kv_cache_0_internal_tensor_assign_27_begin_mask_0, end = concat_237, end_mask = model_model_kv_cache_0_internal_tensor_assign_27_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_27_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_27_stride_0, update = key_states_133, x = coreml_update_state_53)[name = string("model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_27_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_110_write_state")]; + tensor coreml_update_state_54 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_110")]; + tensor expand_dims_162 = const()[name = string("expand_dims_162"), val = tensor([55])]; + tensor expand_dims_163 = const()[name = string("expand_dims_163"), val = tensor([0])]; + tensor expand_dims_165 = const()[name = string("expand_dims_165"), val = tensor([0])]; + tensor expand_dims_166 = const()[name = string("expand_dims_166"), val = tensor([56])]; + int32 concat_240_axis_0 = const()[name = string("concat_240_axis_0"), val = int32(0)]; + bool concat_240_interleave_0 = const()[name = string("concat_240_interleave_0"), val = bool(false)]; + tensor concat_240 = concat(axis = concat_240_axis_0, interleave = concat_240_interleave_0, values = (expand_dims_162, expand_dims_163, current_pos, expand_dims_165))[name = string("concat_240")]; + tensor concat_241_values1_0 = const()[name = string("concat_241_values1_0"), val = tensor([0])]; + tensor concat_241_values3_0 = const()[name = string("concat_241_values3_0"), val = tensor([0])]; + int32 concat_241_axis_0 = const()[name = string("concat_241_axis_0"), val = int32(0)]; + bool concat_241_interleave_0 = const()[name = string("concat_241_interleave_0"), val = bool(false)]; + tensor concat_241 = concat(axis = concat_241_axis_0, interleave = concat_241_interleave_0, values = (expand_dims_166, concat_241_values1_0, var_1042, concat_241_values3_0))[name = string("concat_241")]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_stride_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; + tensor value_states_107 = transpose(perm = var_7923, x = var_7918)[name = string("transpose_5")]; + tensor model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_240, begin_mask = model_model_kv_cache_0_internal_tensor_assign_28_begin_mask_0, end = concat_241, end_mask = model_model_kv_cache_0_internal_tensor_assign_28_end_mask_0, squeeze_mask = model_model_kv_cache_0_internal_tensor_assign_28_squeeze_mask_0, stride = model_model_kv_cache_0_internal_tensor_assign_28_stride_0, update = value_states_107, x = coreml_update_state_54)[name = string("model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16")]; + write_state(data = model_model_kv_cache_0_internal_tensor_assign_28_cast_fp16, input = model_model_kv_cache_0)[name = string("coreml_update_state_111_write_state")]; + tensor coreml_update_state_55 = read_state(input = model_model_kv_cache_0)[name = string("coreml_update_state_111")]; + tensor var_8111_begin_0 = const()[name = string("op_8111_begin_0"), val = tensor([27, 0, 0, 0])]; + tensor var_8111_end_0 = const()[name = string("op_8111_end_0"), val = tensor([28, 8, 1024, 128])]; + tensor var_8111_end_mask_0 = const()[name = string("op_8111_end_mask_0"), val = tensor([false, true, true, true])]; + tensor var_8111_cast_fp16 = slice_by_index(begin = var_8111_begin_0, end = var_8111_end_0, end_mask = var_8111_end_mask_0, x = coreml_update_state_55)[name = string("op_8111_cast_fp16")]; + tensor K_layer_cache_axes_0 = const()[name = string("K_layer_cache_axes_0"), val = tensor([0])]; + tensor K_layer_cache_cast_fp16 = squeeze(axes = K_layer_cache_axes_0, x = var_8111_cast_fp16)[name = string("K_layer_cache_cast_fp16")]; + tensor var_8118_begin_0 = const()[name = string("op_8118_begin_0"), val = tensor([55, 0, 0, 0])]; + tensor var_8118_end_0 = const()[name = string("op_8118_end_0"), val = tensor([1, 8, 1024, 128])]; + tensor var_8118_end_mask_0 = const()[name = string("op_8118_end_mask_0"), val = tensor([true, true, true, true])]; + tensor var_8118_cast_fp16 = slice_by_index(begin = var_8118_begin_0, end = var_8118_end_0, end_mask = var_8118_end_mask_0, x = coreml_update_state_55)[name = string("op_8118_cast_fp16")]; + tensor V_layer_cache_axes_0 = const()[name = string("V_layer_cache_axes_0"), val = tensor([0])]; + tensor V_layer_cache_cast_fp16 = squeeze(axes = V_layer_cache_axes_0, x = var_8118_cast_fp16)[name = string("V_layer_cache_cast_fp16")]; + tensor x_211_axes_0 = const()[name = string("x_211_axes_0"), val = tensor([1])]; + tensor x_211_cast_fp16 = expand_dims(axes = x_211_axes_0, x = K_layer_cache_cast_fp16)[name = string("x_211_cast_fp16")]; + tensor var_8147 = const()[name = string("op_8147"), val = tensor([1, 2, 1, 1])]; + tensor x_213_cast_fp16 = tile(reps = var_8147, x = x_211_cast_fp16)[name = string("x_213_cast_fp16")]; + tensor var_8159 = const()[name = string("op_8159"), val = tensor([1, -1, 1024, 128])]; + tensor key_states_137_cast_fp16 = reshape(shape = var_8159, x = x_213_cast_fp16)[name = string("key_states_137_cast_fp16")]; + tensor x_217_axes_0 = const()[name = string("x_217_axes_0"), val = tensor([1])]; + tensor x_217_cast_fp16 = expand_dims(axes = x_217_axes_0, x = V_layer_cache_cast_fp16)[name = string("x_217_cast_fp16")]; + tensor var_8167 = const()[name = string("op_8167"), val = tensor([1, 2, 1, 1])]; + tensor x_219_cast_fp16 = tile(reps = var_8167, x = x_217_cast_fp16)[name = string("x_219_cast_fp16")]; + bool var_8194_transpose_x_0 = const()[name = string("op_8194_transpose_x_0"), val = bool(false)]; + bool var_8194_transpose_y_0 = const()[name = string("op_8194_transpose_y_0"), val = bool(true)]; + tensor var_8194 = matmul(transpose_x = var_8194_transpose_x_0, transpose_y = var_8194_transpose_y_0, x = query_states_107, y = key_states_137_cast_fp16)[name = string("op_8194")]; + fp16 var_8195_to_fp16 = const()[name = string("op_8195_to_fp16"), val = fp16(0x1.6ap-4)]; + tensor attn_weights_53_cast_fp16 = mul(x = var_8194, y = var_8195_to_fp16)[name = string("attn_weights_53_cast_fp16")]; + tensor attn_weights_cast_fp16 = add(x = attn_weights_53_cast_fp16, y = causal_mask)[name = string("attn_weights_cast_fp16")]; + int32 var_8230 = const()[name = string("op_8230"), val = int32(-1)]; + tensor var_8232_cast_fp16 = softmax(axis = var_8230, x = attn_weights_cast_fp16)[name = string("op_8232_cast_fp16")]; + tensor concat_246 = const()[name = string("concat_246"), val = tensor([16, 128, 1024])]; + tensor reshape_39_cast_fp16 = reshape(shape = concat_246, x = var_8232_cast_fp16)[name = string("reshape_39_cast_fp16")]; + tensor concat_247 = const()[name = string("concat_247"), val = tensor([16, 1024, 128])]; + tensor reshape_40_cast_fp16 = reshape(shape = concat_247, x = x_219_cast_fp16)[name = string("reshape_40_cast_fp16")]; + bool matmul_13_transpose_x_0 = const()[name = string("matmul_13_transpose_x_0"), val = bool(false)]; + bool matmul_13_transpose_y_0 = const()[name = string("matmul_13_transpose_y_0"), val = bool(false)]; + tensor matmul_13_cast_fp16 = matmul(transpose_x = matmul_13_transpose_x_0, transpose_y = matmul_13_transpose_y_0, x = reshape_39_cast_fp16, y = reshape_40_cast_fp16)[name = string("matmul_13_cast_fp16")]; + tensor concat_251 = const()[name = string("concat_251"), val = tensor([1, 16, 128, 128])]; + tensor reshape_41_cast_fp16 = reshape(shape = concat_251, x = matmul_13_cast_fp16)[name = string("reshape_41_cast_fp16")]; + tensor var_8244_perm_0 = const()[name = string("op_8244_perm_0"), val = tensor([0, 2, 1, 3])]; + tensor var_8263 = const()[name = string("op_8263"), val = tensor([1, 128, 2048])]; + tensor var_8244_cast_fp16 = transpose(perm = var_8244_perm_0, x = reshape_41_cast_fp16)[name = string("transpose_4")]; + tensor attn_output_135_cast_fp16 = reshape(shape = var_8263, x = var_8244_cast_fp16)[name = string("attn_output_135_cast_fp16")]; + tensor var_8268 = const()[name = string("op_8268"), val = tensor([0, 2, 1])]; + string var_8284_pad_type_0 = const()[name = string("op_8284_pad_type_0"), val = string("valid")]; + int32 var_8284_groups_0 = const()[name = string("op_8284_groups_0"), val = int32(1)]; + tensor var_8284_strides_0 = const()[name = string("op_8284_strides_0"), val = tensor([1])]; + tensor var_8284_pad_0 = const()[name = string("op_8284_pad_0"), val = tensor([0, 0])]; + tensor var_8284_dilations_0 = const()[name = string("op_8284_dilations_0"), val = tensor([1])]; + tensor squeeze_13_cast_fp16_to_fp32_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719850240))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724044608))))[name = string("squeeze_13_cast_fp16_to_fp32_to_fp16_palettized")]; + tensor var_8269_cast_fp16 = transpose(perm = var_8268, x = attn_output_135_cast_fp16)[name = string("transpose_3")]; + tensor var_8284_cast_fp16 = conv(dilations = var_8284_dilations_0, groups = var_8284_groups_0, pad = var_8284_pad_0, pad_type = var_8284_pad_type_0, strides = var_8284_strides_0, weight = squeeze_13_cast_fp16_to_fp32_to_fp16_palettized, x = var_8269_cast_fp16)[name = string("op_8284_cast_fp16")]; + tensor var_8288 = const()[name = string("op_8288"), val = tensor([0, 2, 1])]; + tensor attn_output_cast_fp16 = transpose(perm = var_8288, x = var_8284_cast_fp16)[name = string("transpose_2")]; + tensor hidden_states_139_cast_fp16 = add(x = hidden_states_131_cast_fp16, y = attn_output_cast_fp16)[name = string("hidden_states_139_cast_fp16")]; + int32 var_8301 = const()[name = string("op_8301"), val = int32(-1)]; + fp16 const_473_promoted_to_fp16 = const()[name = string("const_473_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8303_cast_fp16 = mul(x = hidden_states_139_cast_fp16, y = const_473_promoted_to_fp16)[name = string("op_8303_cast_fp16")]; + bool input_245_interleave_0 = const()[name = string("input_245_interleave_0"), val = bool(false)]; + tensor input_245_cast_fp16 = concat(axis = var_8301, interleave = input_245_interleave_0, values = (hidden_states_139_cast_fp16, var_8303_cast_fp16))[name = string("input_245_cast_fp16")]; + tensor normed_221_axes_0 = const()[name = string("normed_221_axes_0"), val = tensor([-1])]; + fp16 var_8298_to_fp16 = const()[name = string("op_8298_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_221_cast_fp16 = layer_norm(axes = normed_221_axes_0, epsilon = var_8298_to_fp16, x = input_245_cast_fp16)[name = string("normed_221_cast_fp16")]; + tensor normed_223_begin_0 = const()[name = string("normed_223_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_223_end_0 = const()[name = string("normed_223_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_223_end_mask_0 = const()[name = string("normed_223_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_223_cast_fp16 = slice_by_index(begin = normed_223_begin_0, end = normed_223_end_0, end_mask = normed_223_end_mask_0, x = normed_221_cast_fp16)[name = string("normed_223_cast_fp16")]; + tensor const_476_promoted_to_fp16 = const()[name = string("const_476_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724175744)))]; + tensor x_221_cast_fp16 = mul(x = normed_223_cast_fp16, y = const_476_promoted_to_fp16)[name = string("x_221_cast_fp16")]; + tensor var_8328 = const()[name = string("op_8328"), val = tensor([0, 2, 1])]; + tensor input_247_axes_0 = const()[name = string("input_247_axes_0"), val = tensor([2])]; + tensor var_8329 = transpose(perm = var_8328, x = x_221_cast_fp16)[name = string("transpose_1")]; + tensor input_247 = expand_dims(axes = input_247_axes_0, x = var_8329)[name = string("input_247")]; + string input_249_pad_type_0 = const()[name = string("input_249_pad_type_0"), val = string("valid")]; + tensor input_249_strides_0 = const()[name = string("input_249_strides_0"), val = tensor([1, 1])]; + tensor input_249_pad_0 = const()[name = string("input_249_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor input_249_dilations_0 = const()[name = string("input_249_dilations_0"), val = tensor([1, 1])]; + int32 input_249_groups_0 = const()[name = string("input_249_groups_0"), val = int32(1)]; + tensor input_249 = conv(dilations = input_249_dilations_0, groups = input_249_groups_0, pad = input_249_pad_0, pad_type = input_249_pad_type_0, strides = input_249_strides_0, weight = model_model_layers_27_mlp_gate_proj_weight_palettized, x = input_247)[name = string("input_249")]; + string b_pad_type_0 = const()[name = string("b_pad_type_0"), val = string("valid")]; + tensor b_strides_0 = const()[name = string("b_strides_0"), val = tensor([1, 1])]; + tensor b_pad_0 = const()[name = string("b_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor b_dilations_0 = const()[name = string("b_dilations_0"), val = tensor([1, 1])]; + int32 b_groups_0 = const()[name = string("b_groups_0"), val = int32(1)]; + tensor b = conv(dilations = b_dilations_0, groups = b_groups_0, pad = b_pad_0, pad_type = b_pad_type_0, strides = b_strides_0, weight = model_model_layers_27_mlp_up_proj_weight_palettized, x = input_247)[name = string("b")]; + tensor c = silu(x = input_249)[name = string("c")]; + tensor input_251 = mul(x = c, y = b)[name = string("input_251")]; + string e_pad_type_0 = const()[name = string("e_pad_type_0"), val = string("valid")]; + tensor e_strides_0 = const()[name = string("e_strides_0"), val = tensor([1, 1])]; + tensor e_pad_0 = const()[name = string("e_pad_0"), val = tensor([0, 0, 0, 0])]; + tensor e_dilations_0 = const()[name = string("e_dilations_0"), val = tensor([1, 1])]; + int32 e_groups_0 = const()[name = string("e_groups_0"), val = int32(1)]; + tensor e = conv(dilations = e_dilations_0, groups = e_groups_0, pad = e_pad_0, pad_type = e_pad_type_0, strides = e_strides_0, weight = model_model_layers_27_mlp_down_proj_weight_palettized, x = input_251)[name = string("e")]; + tensor var_8351_axes_0 = const()[name = string("op_8351_axes_0"), val = tensor([2])]; + tensor var_8351 = squeeze(axes = var_8351_axes_0, x = e)[name = string("op_8351")]; + tensor var_8352 = const()[name = string("op_8352"), val = tensor([0, 2, 1])]; + tensor var_8353 = transpose(perm = var_8352, x = var_8351)[name = string("transpose_0")]; + tensor hidden_states_cast_fp16 = add(x = hidden_states_139_cast_fp16, y = var_8353)[name = string("hidden_states_cast_fp16")]; + int32 var_8365 = const()[name = string("op_8365"), val = int32(-1)]; + fp16 const_477_promoted_to_fp16 = const()[name = string("const_477_promoted_to_fp16"), val = fp16(-0x1p+0)]; + tensor var_8367_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_477_promoted_to_fp16)[name = string("op_8367_cast_fp16")]; + bool input_interleave_0 = const()[name = string("input_interleave_0"), val = bool(false)]; + tensor input_cast_fp16 = concat(axis = var_8365, interleave = input_interleave_0, values = (hidden_states_cast_fp16, var_8367_cast_fp16))[name = string("input_cast_fp16")]; + tensor normed_225_axes_0 = const()[name = string("normed_225_axes_0"), val = tensor([-1])]; + fp16 var_8362_to_fp16 = const()[name = string("op_8362_to_fp16"), val = fp16(0x1.1p-20)]; + tensor normed_225_cast_fp16 = layer_norm(axes = normed_225_axes_0, epsilon = var_8362_to_fp16, x = input_cast_fp16)[name = string("normed_225_cast_fp16")]; + tensor normed_begin_0 = const()[name = string("normed_begin_0"), val = tensor([0, 0, 0])]; + tensor normed_end_0 = const()[name = string("normed_end_0"), val = tensor([1, 128, 2048])]; + tensor normed_end_mask_0 = const()[name = string("normed_end_mask_0"), val = tensor([true, true, false])]; + tensor normed_cast_fp16 = slice_by_index(begin = normed_begin_0, end = normed_end_0, end_mask = normed_end_mask_0, x = normed_225_cast_fp16)[name = string("normed_cast_fp16")]; + tensor const_480_promoted_to_fp16 = const()[name = string("const_480_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724179904)))]; + tensor output_hidden_states = mul(x = normed_cast_fp16, y = const_480_promoted_to_fp16)[name = string("op_8380_cast_fp16")]; + } -> (output_hidden_states); +} \ No newline at end of file