program(1.3) [buildInfo = dict({{"coremlc-component-MIL", "3500.14.1"}, {"coremlc-version", "3500.32.1"}})] { func infer(tensor causal_mask_full, tensor causal_mask_sliding, tensor cos_f, tensor cos_s, tensor current_pos, tensor hidden_states, state> kv_cache_full, state> kv_cache_sliding, tensor per_layer_raw, tensor ring_pos, tensor sin_f, tensor sin_s) { tensor per_layer_model_projection_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6881408))))[name = string("per_layer_model_projection_weight_palettized")]; tensor layers_0_input_layernorm_weight = const()[name = string("layers_0_input_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6890432)))]; tensor layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6893568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8466496))))[name = string("layers_0_self_attn_q_proj_weight_palettized")]; tensor layers_0_self_attn_q_norm_weight = const()[name = string("layers_0_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8468608)))]; tensor layers_0_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8469184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8665856))))[name = string("layers_0_self_attn_k_proj_weight_palettized")]; tensor layers_0_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8666176))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8862848))))[name = string("layers_0_self_attn_v_proj_weight_palettized")]; tensor layers_0_self_attn_k_norm_weight = const()[name = string("layers_0_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8863168)))]; tensor layers_0_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8863744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13582400))))[name = string("layers_0_mlp_gate_proj_weight_palettized")]; tensor layers_0_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13588608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18307264))))[name = string("layers_0_mlp_up_proj_weight_palettized")]; tensor layers_0_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18313472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23032128))))[name = string("layers_0_mlp_down_proj_weight_palettized")]; tensor layers_0_post_feedforward_layernorm_weight = const()[name = string("layers_0_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23033728)))]; tensor layers_0_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23036864))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23233536))))[name = string("layers_0_per_layer_input_gate_weight_palettized")]; tensor layers_1_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23233856))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24806784))))[name = string("layers_1_self_attn_q_proj_weight_palettized")]; tensor layers_1_self_attn_q_norm_weight = const()[name = string("layers_1_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24808896)))]; tensor layers_1_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24809472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25006144))))[name = string("layers_1_self_attn_k_proj_weight_palettized")]; tensor layers_1_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25006464))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25203136))))[name = string("layers_1_self_attn_v_proj_weight_palettized")]; tensor layers_1_self_attn_k_norm_weight = const()[name = string("layers_1_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25203456)))]; tensor layers_1_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25204032))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(29922688))))[name = string("layers_1_mlp_gate_proj_weight_palettized")]; tensor layers_1_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(29928896))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34647552))))[name = string("layers_1_mlp_up_proj_weight_palettized")]; tensor layers_1_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34653760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39372416))))[name = string("layers_1_mlp_down_proj_weight_palettized")]; tensor layers_1_post_feedforward_layernorm_weight = const()[name = string("layers_1_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39374016)))]; tensor layers_1_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39377152))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39573824))))[name = string("layers_1_per_layer_input_gate_weight_palettized")]; tensor layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39574144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41147072))))[name = string("layers_2_self_attn_q_proj_weight_palettized")]; tensor layers_2_self_attn_q_norm_weight = const()[name = string("layers_2_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41149184)))]; tensor layers_2_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41149760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41346432))))[name = string("layers_2_self_attn_k_proj_weight_palettized")]; tensor layers_2_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41346752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41543424))))[name = string("layers_2_self_attn_v_proj_weight_palettized")]; tensor layers_2_self_attn_k_norm_weight = const()[name = string("layers_2_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41543744)))]; tensor layers_2_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41544320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46262976))))[name = string("layers_2_mlp_gate_proj_weight_palettized")]; tensor layers_2_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46269184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50987840))))[name = string("layers_2_mlp_up_proj_weight_palettized")]; tensor layers_2_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50994048))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55712704))))[name = string("layers_2_mlp_down_proj_weight_palettized")]; tensor layers_2_post_feedforward_layernorm_weight = const()[name = string("layers_2_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55714304)))]; tensor layers_2_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55717440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55914112))))[name = string("layers_2_per_layer_input_gate_weight_palettized")]; tensor layers_3_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55914432))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57487360))))[name = string("layers_3_self_attn_q_proj_weight_palettized")]; tensor layers_3_self_attn_q_norm_weight = const()[name = string("layers_3_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57489472)))]; tensor layers_3_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57490048))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57686720))))[name = string("layers_3_self_attn_k_proj_weight_palettized")]; tensor layers_3_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57687040))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57883712))))[name = string("layers_3_self_attn_v_proj_weight_palettized")]; tensor layers_3_self_attn_k_norm_weight = const()[name = string("layers_3_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57884032)))]; tensor layers_3_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57884608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62603264))))[name = string("layers_3_mlp_gate_proj_weight_palettized")]; tensor layers_3_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62609472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67328128))))[name = string("layers_3_mlp_up_proj_weight_palettized")]; tensor layers_3_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67334336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72052992))))[name = string("layers_3_mlp_down_proj_weight_palettized")]; tensor layers_3_post_feedforward_layernorm_weight = const()[name = string("layers_3_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72054592)))]; tensor layers_3_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72057728))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72254400))))[name = string("layers_3_per_layer_input_gate_weight_palettized")]; tensor layers_4_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72254720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75400512))))[name = string("layers_4_self_attn_q_proj_weight_palettized")]; tensor layers_4_self_attn_q_norm_weight = const()[name = string("layers_4_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75404672)))]; tensor layers_4_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75405760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75799040))))[name = string("layers_4_self_attn_k_proj_weight_palettized")]; tensor layers_4_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75799616))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(76192896))))[name = string("layers_4_self_attn_v_proj_weight_palettized")]; tensor layers_4_self_attn_k_norm_weight = const()[name = string("layers_4_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(76193472)))]; tensor layers_4_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(76194560))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80913216))))[name = string("layers_4_mlp_gate_proj_weight_palettized")]; tensor layers_4_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80919424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85638080))))[name = string("layers_4_mlp_up_proj_weight_palettized")]; tensor layers_4_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85644288))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90362944))))[name = string("layers_4_mlp_down_proj_weight_palettized")]; tensor layers_4_post_feedforward_layernorm_weight = const()[name = string("layers_4_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90364544)))]; tensor layers_4_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90367680))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90564352))))[name = string("layers_4_per_layer_input_gate_weight_palettized")]; tensor layers_5_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90564672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(92137600))))[name = string("layers_5_self_attn_q_proj_weight_palettized")]; tensor layers_5_self_attn_q_norm_weight = const()[name = string("layers_5_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(92139712)))]; tensor layers_5_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(92140288))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(92336960))))[name = string("layers_5_self_attn_k_proj_weight_palettized")]; tensor layers_5_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(92337280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(92533952))))[name = string("layers_5_self_attn_v_proj_weight_palettized")]; tensor layers_5_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(92534272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(97252928))))[name = string("layers_5_mlp_gate_proj_weight_palettized")]; tensor layers_5_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(97259136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101977792))))[name = string("layers_5_mlp_up_proj_weight_palettized")]; tensor layers_5_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101984000))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(106702656))))[name = string("layers_5_mlp_down_proj_weight_palettized")]; tensor layers_5_post_feedforward_layernorm_weight = const()[name = string("layers_5_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(106704256)))]; tensor layers_5_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(106707392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(106904064))))[name = string("layers_5_per_layer_input_gate_weight_palettized")]; tensor layers_6_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(106904384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108477312))))[name = string("layers_6_self_attn_q_proj_weight_palettized")]; tensor layers_6_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108479424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108676096))))[name = string("layers_6_self_attn_k_proj_weight_palettized")]; tensor layers_6_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108676416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108873088))))[name = string("layers_6_self_attn_v_proj_weight_palettized")]; tensor layers_6_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108873408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(113592064))))[name = string("layers_6_mlp_gate_proj_weight_palettized")]; tensor layers_6_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(113598272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(118316928))))[name = string("layers_6_mlp_up_proj_weight_palettized")]; tensor layers_6_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(118323136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(123041792))))[name = string("layers_6_mlp_down_proj_weight_palettized")]; tensor layers_6_post_feedforward_layernorm_weight = const()[name = string("layers_6_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(123043392)))]; tensor layers_6_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(123046528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(123243200))))[name = string("layers_6_per_layer_input_gate_weight_palettized")]; tensor layers_7_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(123243520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(124816448))))[name = string("layers_7_self_attn_q_proj_weight_palettized")]; tensor layers_7_self_attn_q_norm_weight = const()[name = string("layers_7_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(124818560)))]; tensor layers_7_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(124819136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125015808))))[name = string("layers_7_self_attn_k_proj_weight_palettized")]; tensor layers_7_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125016128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125212800))))[name = string("layers_7_self_attn_v_proj_weight_palettized")]; tensor layers_7_self_attn_k_norm_weight = const()[name = string("layers_7_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125213120)))]; tensor layers_7_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125213696))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(129932352))))[name = string("layers_7_mlp_gate_proj_weight_palettized")]; tensor layers_7_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(129938560))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(134657216))))[name = string("layers_7_mlp_up_proj_weight_palettized")]; tensor layers_7_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(134663424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139382080))))[name = string("layers_7_mlp_down_proj_weight_palettized")]; tensor layers_7_post_feedforward_layernorm_weight = const()[name = string("layers_7_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139383680)))]; tensor layers_7_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139386816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139583488))))[name = string("layers_7_per_layer_input_gate_weight_palettized")]; tensor linear_0_bias_0 = const()[name = string("linear_0_bias_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139583808)))]; tensor var_519 = linear(bias = linear_0_bias_0, weight = per_layer_model_projection_weight_palettized, x = hidden_states)[name = string("linear_0")]; fp16 var_520_to_fp16 = const()[name = string("op_520_to_fp16"), val = fp16(0x1.a2p-6)]; tensor proj_cast_fp16 = mul(x = var_519, y = var_520_to_fp16)[name = string("proj_cast_fp16")]; tensor var_525 = const()[name = string("op_525"), val = tensor([1, 35, 256])]; tensor proj_grouped_cast_fp16 = reshape(shape = var_525, x = proj_cast_fp16)[name = string("proj_grouped_cast_fp16")]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_527_cast_fp16 = mul(x = proj_grouped_cast_fp16, y = const_0_promoted_to_fp16)[name = string("op_527_cast_fp16")]; int32 var_529 = const()[name = string("op_529"), val = int32(-1)]; bool input_3_interleave_0 = const()[name = string("input_3_interleave_0"), val = bool(false)]; tensor input_3_cast_fp16 = concat(axis = var_529, interleave = input_3_interleave_0, values = (proj_grouped_cast_fp16, var_527_cast_fp16))[name = string("input_3_cast_fp16")]; tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; fp16 var_535_to_fp16 = const()[name = string("op_535_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_535_to_fp16, x = input_3_cast_fp16)[name = string("normed_1_cast_fp16")]; tensor var_538_split_sizes_0 = const()[name = string("op_538_split_sizes_0"), val = tensor([256, 256])]; int32 var_538_axis_0 = const()[name = string("op_538_axis_0"), val = int32(-1)]; tensor var_538_cast_fp16_0, tensor var_538_cast_fp16_1 = split(axis = var_538_axis_0, split_sizes = var_538_split_sizes_0, x = normed_1_cast_fp16)[name = string("op_538_cast_fp16")]; tensor per_layer_projection_norm_weight_promoted_to_fp16 = const()[name = string("per_layer_projection_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139601792)))]; tensor var_540_cast_fp16 = mul(x = var_538_cast_fp16_0, y = per_layer_projection_norm_weight_promoted_to_fp16)[name = string("op_540_cast_fp16")]; tensor var_544 = const()[name = string("op_544"), val = tensor([1, 1, 8960])]; tensor proj_normed_cast_fp16 = reshape(shape = var_544, x = var_540_cast_fp16)[name = string("proj_normed_cast_fp16")]; tensor var_547_cast_fp16 = add(x = proj_normed_cast_fp16, y = per_layer_raw)[name = string("op_547_cast_fp16")]; fp16 var_548_to_fp16 = const()[name = string("op_548_to_fp16"), val = fp16(0x1.6ap-1)]; tensor per_layer_combined_out = mul(x = var_547_cast_fp16, y = var_548_to_fp16)[name = string("per_layer_combined_cast_fp16")]; int32 var_554 = const()[name = string("op_554"), val = int32(-1)]; fp16 const_1_promoted = const()[name = string("const_1_promoted"), val = fp16(-0x1p+0)]; tensor var_556 = mul(x = hidden_states, y = const_1_promoted)[name = string("op_556")]; bool input_5_interleave_0 = const()[name = string("input_5_interleave_0"), val = bool(false)]; tensor input_5 = concat(axis = var_554, interleave = input_5_interleave_0, values = (hidden_states, var_556))[name = string("input_5")]; tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; fp16 var_551_to_fp16 = const()[name = string("op_551_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_551_to_fp16, x = input_5)[name = string("normed_5_cast_fp16")]; tensor var_561_split_sizes_0 = const()[name = string("op_561_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_561_axis_0 = const()[name = string("op_561_axis_0"), val = int32(-1)]; tensor var_561_0, tensor var_561_1 = split(axis = var_561_axis_0, split_sizes = var_561_split_sizes_0, x = normed_5_cast_fp16)[name = string("op_561")]; tensor var_563 = mul(x = var_561_0, y = layers_0_input_layernorm_weight)[name = string("op_563")]; tensor linear_1_bias_0 = const()[name = string("linear_1_bias_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139602368)))]; tensor var_571 = linear(bias = linear_1_bias_0, weight = layers_0_self_attn_q_proj_weight_palettized, x = var_563)[name = string("linear_1")]; tensor var_576 = const()[name = string("op_576"), val = tensor([1, 8, 256, 1])]; tensor var_577 = reshape(shape = var_576, x = var_571)[name = string("op_577")]; tensor var_582 = const()[name = string("op_582"), val = tensor([0, 1, 3, 2])]; tensor var_592 = const()[name = string("op_592"), val = tensor([1, 8, 256])]; tensor var_583 = transpose(perm = var_582, x = var_577)[name = string("transpose_79")]; tensor x_1 = reshape(shape = var_592, x = var_583)[name = string("x_1")]; int32 var_598 = const()[name = string("op_598"), val = int32(-1)]; fp16 const_2_promoted = const()[name = string("const_2_promoted"), val = fp16(-0x1p+0)]; tensor var_600 = mul(x = x_1, y = const_2_promoted)[name = string("op_600")]; bool input_9_interleave_0 = const()[name = string("input_9_interleave_0"), val = bool(false)]; tensor input_9 = concat(axis = var_598, interleave = input_9_interleave_0, values = (x_1, var_600))[name = string("input_9")]; tensor normed_9_axes_0 = const()[name = string("normed_9_axes_0"), val = tensor([-1])]; fp16 var_595_to_fp16 = const()[name = string("op_595_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_9_cast_fp16 = layer_norm(axes = normed_9_axes_0, epsilon = var_595_to_fp16, x = input_9)[name = string("normed_9_cast_fp16")]; tensor var_605_split_sizes_0 = const()[name = string("op_605_split_sizes_0"), val = tensor([256, 256])]; int32 var_605_axis_0 = const()[name = string("op_605_axis_0"), val = int32(-1)]; tensor var_605_0, tensor var_605_1 = split(axis = var_605_axis_0, split_sizes = var_605_split_sizes_0, x = normed_9_cast_fp16)[name = string("op_605")]; tensor var_607 = mul(x = var_605_0, y = layers_0_self_attn_q_norm_weight)[name = string("op_607")]; tensor var_612 = const()[name = string("op_612"), val = tensor([1, 8, 1, 256])]; tensor q_3 = reshape(shape = var_612, x = var_607)[name = string("q_3")]; tensor var_614_cast_fp16 = mul(x = q_3, y = cos_s)[name = string("op_614_cast_fp16")]; tensor var_615_split_sizes_0 = const()[name = string("op_615_split_sizes_0"), val = tensor([128, 128])]; int32 var_615_axis_0 = const()[name = string("op_615_axis_0"), val = int32(-1)]; tensor var_615_0, tensor var_615_1 = split(axis = var_615_axis_0, split_sizes = var_615_split_sizes_0, x = q_3)[name = string("op_615")]; fp16 const_3_promoted = const()[name = string("const_3_promoted"), val = fp16(-0x1p+0)]; tensor var_617 = mul(x = var_615_1, y = const_3_promoted)[name = string("op_617")]; int32 var_619 = const()[name = string("op_619"), val = int32(-1)]; bool var_620_interleave_0 = const()[name = string("op_620_interleave_0"), val = bool(false)]; tensor var_620 = concat(axis = var_619, interleave = var_620_interleave_0, values = (var_617, var_615_0))[name = string("op_620")]; tensor var_621_cast_fp16 = mul(x = var_620, y = sin_s)[name = string("op_621_cast_fp16")]; tensor q_7_cast_fp16 = add(x = var_614_cast_fp16, y = var_621_cast_fp16)[name = string("q_7_cast_fp16")]; tensor linear_2_bias_0 = const()[name = string("linear_2_bias_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139606528)))]; tensor var_626 = linear(bias = linear_2_bias_0, weight = layers_0_self_attn_k_proj_weight_palettized, x = var_563)[name = string("linear_2")]; tensor var_631 = const()[name = string("op_631"), val = tensor([1, 1, 256, 1])]; tensor var_632 = reshape(shape = var_631, x = var_626)[name = string("op_632")]; tensor var_637 = const()[name = string("op_637"), val = tensor([0, 1, 3, 2])]; tensor var_646 = linear(bias = linear_2_bias_0, weight = layers_0_self_attn_v_proj_weight_palettized, x = var_563)[name = string("linear_3")]; tensor var_651 = const()[name = string("op_651"), val = tensor([1, 1, 256, 1])]; tensor var_652 = reshape(shape = var_651, x = var_646)[name = string("op_652")]; tensor var_657 = const()[name = string("op_657"), val = tensor([0, 1, 3, 2])]; tensor var_667 = const()[name = string("op_667"), val = tensor([1, 1, 256])]; tensor var_638 = transpose(perm = var_637, x = var_632)[name = string("transpose_78")]; tensor x_3 = reshape(shape = var_667, x = var_638)[name = string("x_3")]; int32 var_673 = const()[name = string("op_673"), val = int32(-1)]; fp16 const_4_promoted = const()[name = string("const_4_promoted"), val = fp16(-0x1p+0)]; tensor var_675 = mul(x = x_3, y = const_4_promoted)[name = string("op_675")]; bool input_11_interleave_0 = const()[name = string("input_11_interleave_0"), val = bool(false)]; tensor input_11 = concat(axis = var_673, interleave = input_11_interleave_0, values = (x_3, var_675))[name = string("input_11")]; tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; fp16 var_670_to_fp16 = const()[name = string("op_670_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_670_to_fp16, x = input_11)[name = string("normed_13_cast_fp16")]; tensor var_680_split_sizes_0 = const()[name = string("op_680_split_sizes_0"), val = tensor([256, 256])]; int32 var_680_axis_0 = const()[name = string("op_680_axis_0"), val = int32(-1)]; tensor var_680_0, tensor var_680_1 = split(axis = var_680_axis_0, split_sizes = var_680_split_sizes_0, x = normed_13_cast_fp16)[name = string("op_680")]; tensor var_682 = mul(x = var_680_0, y = layers_0_self_attn_k_norm_weight)[name = string("op_682")]; tensor var_687 = const()[name = string("op_687"), val = tensor([1, 1, 1, 256])]; tensor q_5 = reshape(shape = var_687, x = var_682)[name = string("q_5")]; fp16 var_689_promoted = const()[name = string("op_689_promoted"), val = fp16(0x1p+1)]; tensor var_658 = transpose(perm = var_657, x = var_652)[name = string("transpose_77")]; tensor var_690 = pow(x = var_658, y = var_689_promoted)[name = string("op_690")]; tensor var_695_axes_0 = const()[name = string("op_695_axes_0"), val = tensor([-1])]; bool var_695_keep_dims_0 = const()[name = string("op_695_keep_dims_0"), val = bool(true)]; tensor var_695 = reduce_mean(axes = var_695_axes_0, keep_dims = var_695_keep_dims_0, x = var_690)[name = string("op_695")]; fp16 var_697_to_fp16 = const()[name = string("op_697_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_1_cast_fp16 = add(x = var_695, y = var_697_to_fp16)[name = string("mean_sq_1_cast_fp16")]; fp32 var_699_epsilon_0 = const()[name = string("op_699_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_699_cast_fp16 = rsqrt(epsilon = var_699_epsilon_0, x = mean_sq_1_cast_fp16)[name = string("op_699_cast_fp16")]; tensor input_15_cast_fp16 = mul(x = var_658, y = var_699_cast_fp16)[name = string("input_15_cast_fp16")]; tensor var_701_cast_fp16 = mul(x = q_5, y = cos_s)[name = string("op_701_cast_fp16")]; tensor var_702_split_sizes_0 = const()[name = string("op_702_split_sizes_0"), val = tensor([128, 128])]; int32 var_702_axis_0 = const()[name = string("op_702_axis_0"), val = int32(-1)]; tensor var_702_0, tensor var_702_1 = split(axis = var_702_axis_0, split_sizes = var_702_split_sizes_0, x = q_5)[name = string("op_702")]; fp16 const_5_promoted = const()[name = string("const_5_promoted"), val = fp16(-0x1p+0)]; tensor var_704 = mul(x = var_702_1, y = const_5_promoted)[name = string("op_704")]; int32 var_706 = const()[name = string("op_706"), val = int32(-1)]; bool var_707_interleave_0 = const()[name = string("op_707_interleave_0"), val = bool(false)]; tensor var_707 = concat(axis = var_706, interleave = var_707_interleave_0, values = (var_704, var_702_0))[name = string("op_707")]; tensor var_708_cast_fp16 = mul(x = var_707, y = sin_s)[name = string("op_708_cast_fp16")]; tensor input_13_cast_fp16 = add(x = var_701_cast_fp16, y = var_708_cast_fp16)[name = string("input_13_cast_fp16")]; tensor k_padded_1_pad_0 = const()[name = string("k_padded_1_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_1_mode_0 = const()[name = string("k_padded_1_mode_0"), val = string("constant")]; fp16 const_6_to_fp16 = const()[name = string("const_6_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_1_cast_fp16 = pad(constant_val = const_6_to_fp16, mode = k_padded_1_mode_0, pad = k_padded_1_pad_0, x = input_13_cast_fp16)[name = string("k_padded_1_cast_fp16")]; tensor v_padded_1_pad_0 = const()[name = string("v_padded_1_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_1_mode_0 = const()[name = string("v_padded_1_mode_0"), val = string("constant")]; fp16 const_7_to_fp16 = const()[name = string("const_7_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_1_cast_fp16 = pad(constant_val = const_7_to_fp16, mode = v_padded_1_mode_0, pad = v_padded_1_pad_0, x = input_15_cast_fp16)[name = string("v_padded_1_cast_fp16")]; int32 var_724 = const()[name = string("op_724"), val = int32(1)]; tensor var_725 = add(x = ring_pos, y = var_724)[name = string("op_725")]; tensor read_state_0 = read_state(input = kv_cache_sliding)[name = string("read_state_0")]; tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([0])]; tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([1])]; int32 concat_2_axis_0 = const()[name = string("concat_2_axis_0"), val = int32(0)]; bool concat_2_interleave_0 = const()[name = string("concat_2_interleave_0"), val = bool(false)]; tensor concat_2 = concat(axis = concat_2_axis_0, interleave = concat_2_interleave_0, values = (expand_dims_0, expand_dims_1, ring_pos, expand_dims_3))[name = string("concat_2")]; tensor concat_3_values1_0 = const()[name = string("concat_3_values1_0"), val = tensor([0])]; tensor concat_3_values3_0 = const()[name = string("concat_3_values3_0"), val = tensor([0])]; int32 concat_3_axis_0 = const()[name = string("concat_3_axis_0"), val = int32(0)]; bool concat_3_interleave_0 = const()[name = string("concat_3_interleave_0"), val = bool(false)]; tensor concat_3 = concat(axis = concat_3_axis_0, interleave = concat_3_interleave_0, values = (expand_dims_4, concat_3_values1_0, var_725, concat_3_values3_0))[name = string("concat_3")]; tensor kv_cache_sliding_internal_tensor_assign_1_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_1_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_1_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_2, begin_mask = kv_cache_sliding_internal_tensor_assign_1_begin_mask_0, end = concat_3, end_mask = kv_cache_sliding_internal_tensor_assign_1_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_1_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_1_stride_0, update = k_padded_1_cast_fp16, x = read_state_0)[name = string("kv_cache_sliding_internal_tensor_assign_1_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_1_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_0_write_state")]; tensor coreml_update_state_16 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_0")]; tensor expand_dims_6 = const()[name = string("expand_dims_6"), val = tensor([1])]; tensor expand_dims_7 = const()[name = string("expand_dims_7"), val = tensor([0])]; tensor expand_dims_9 = const()[name = string("expand_dims_9"), val = tensor([0])]; tensor expand_dims_10 = const()[name = string("expand_dims_10"), val = tensor([2])]; int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_6, expand_dims_7, ring_pos, expand_dims_9))[name = string("concat_6")]; tensor concat_7_values1_0 = const()[name = string("concat_7_values1_0"), val = tensor([0])]; tensor concat_7_values3_0 = const()[name = string("concat_7_values3_0"), val = tensor([0])]; int32 concat_7_axis_0 = const()[name = string("concat_7_axis_0"), val = int32(0)]; bool concat_7_interleave_0 = const()[name = string("concat_7_interleave_0"), val = bool(false)]; tensor concat_7 = concat(axis = concat_7_axis_0, interleave = concat_7_interleave_0, values = (expand_dims_10, concat_7_values1_0, var_725, concat_7_values3_0))[name = string("concat_7")]; tensor kv_cache_sliding_internal_tensor_assign_2_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_2_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_2_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_6, begin_mask = kv_cache_sliding_internal_tensor_assign_2_begin_mask_0, end = concat_7, end_mask = kv_cache_sliding_internal_tensor_assign_2_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_2_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_2_stride_0, update = v_padded_1_cast_fp16, x = coreml_update_state_16)[name = string("kv_cache_sliding_internal_tensor_assign_2_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_2_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_1_write_state")]; tensor coreml_update_state_17 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_1")]; tensor var_775_begin_0 = const()[name = string("op_775_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_775_end_0 = const()[name = string("op_775_end_0"), val = tensor([1, 1, 512, 512])]; tensor var_775_end_mask_0 = const()[name = string("op_775_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_775_cast_fp16 = slice_by_index(begin = var_775_begin_0, end = var_775_end_0, end_mask = var_775_end_mask_0, x = coreml_update_state_17)[name = string("op_775_cast_fp16")]; tensor K_sliding_slice_1_begin_0 = const()[name = string("K_sliding_slice_1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_sliding_slice_1_end_0 = const()[name = string("K_sliding_slice_1_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_sliding_slice_1_end_mask_0 = const()[name = string("K_sliding_slice_1_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_sliding_slice_1_cast_fp16 = slice_by_index(begin = K_sliding_slice_1_begin_0, end = K_sliding_slice_1_end_0, end_mask = K_sliding_slice_1_end_mask_0, x = var_775_cast_fp16)[name = string("K_sliding_slice_1_cast_fp16")]; tensor var_795_begin_0 = const()[name = string("op_795_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_795_end_0 = const()[name = string("op_795_end_0"), val = tensor([2, 1, 512, 512])]; tensor var_795_end_mask_0 = const()[name = string("op_795_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_795_cast_fp16 = slice_by_index(begin = var_795_begin_0, end = var_795_end_0, end_mask = var_795_end_mask_0, x = coreml_update_state_17)[name = string("op_795_cast_fp16")]; tensor V_for_attn_1_begin_0 = const()[name = string("V_for_attn_1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_1_end_0 = const()[name = string("V_for_attn_1_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_1_end_mask_0 = const()[name = string("V_for_attn_1_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_1_cast_fp16 = slice_by_index(begin = V_for_attn_1_begin_0, end = V_for_attn_1_end_0, end_mask = V_for_attn_1_end_mask_0, x = var_795_cast_fp16)[name = string("V_for_attn_1_cast_fp16")]; tensor transpose_0_perm_0 = const()[name = string("transpose_0_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_0_reps_0 = const()[name = string("tile_0_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_0_cast_fp16 = transpose(perm = transpose_0_perm_0, x = K_sliding_slice_1_cast_fp16)[name = string("transpose_76")]; tensor tile_0_cast_fp16 = tile(reps = tile_0_reps_0, x = transpose_0_cast_fp16)[name = string("tile_0_cast_fp16")]; tensor concat_8 = const()[name = string("concat_8"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_0_cast_fp16 = reshape(shape = concat_8, x = tile_0_cast_fp16)[name = string("reshape_0_cast_fp16")]; tensor transpose_1_perm_0 = const()[name = string("transpose_1_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_9 = const()[name = string("concat_9"), val = tensor([-1, 1, 512, 256])]; tensor transpose_1_cast_fp16 = transpose(perm = transpose_1_perm_0, x = reshape_0_cast_fp16)[name = string("transpose_75")]; tensor reshape_1_cast_fp16 = reshape(shape = concat_9, x = transpose_1_cast_fp16)[name = string("reshape_1_cast_fp16")]; tensor transpose_32_perm_0 = const()[name = string("transpose_32_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_2_perm_0 = const()[name = string("transpose_2_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_1_reps_0 = const()[name = string("tile_1_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_2_cast_fp16 = transpose(perm = transpose_2_perm_0, x = V_for_attn_1_cast_fp16)[name = string("transpose_74")]; tensor tile_1_cast_fp16 = tile(reps = tile_1_reps_0, x = transpose_2_cast_fp16)[name = string("tile_1_cast_fp16")]; tensor concat_10 = const()[name = string("concat_10"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_2_cast_fp16 = reshape(shape = concat_10, x = tile_1_cast_fp16)[name = string("reshape_2_cast_fp16")]; tensor transpose_3_perm_0 = const()[name = string("transpose_3_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_11 = const()[name = string("concat_11"), val = tensor([-1, 1, 512, 256])]; tensor transpose_3_cast_fp16 = transpose(perm = transpose_3_perm_0, x = reshape_2_cast_fp16)[name = string("transpose_73")]; tensor reshape_3_cast_fp16 = reshape(shape = concat_11, x = transpose_3_cast_fp16)[name = string("reshape_3_cast_fp16")]; tensor V_expanded_1_perm_0 = const()[name = string("V_expanded_1_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_1_transpose_x_0 = const()[name = string("attn_weights_1_transpose_x_0"), val = bool(false)]; bool attn_weights_1_transpose_y_0 = const()[name = string("attn_weights_1_transpose_y_0"), val = bool(false)]; tensor transpose_32_cast_fp16 = transpose(perm = transpose_32_perm_0, x = reshape_1_cast_fp16)[name = string("transpose_72")]; tensor attn_weights_1_cast_fp16 = matmul(transpose_x = attn_weights_1_transpose_x_0, transpose_y = attn_weights_1_transpose_y_0, x = q_7_cast_fp16, y = transpose_32_cast_fp16)[name = string("attn_weights_1_cast_fp16")]; tensor x_7_cast_fp16 = add(x = attn_weights_1_cast_fp16, y = causal_mask_sliding)[name = string("x_7_cast_fp16")]; tensor reduce_max_0_axes_0 = const()[name = string("reduce_max_0_axes_0"), val = tensor([-1])]; bool reduce_max_0_keep_dims_0 = const()[name = string("reduce_max_0_keep_dims_0"), val = bool(true)]; tensor reduce_max_0 = reduce_max(axes = reduce_max_0_axes_0, keep_dims = reduce_max_0_keep_dims_0, x = x_7_cast_fp16)[name = string("reduce_max_0")]; tensor var_840 = sub(x = x_7_cast_fp16, y = reduce_max_0)[name = string("op_840")]; tensor var_846 = exp(x = var_840)[name = string("op_846")]; tensor var_856_axes_0 = const()[name = string("op_856_axes_0"), val = tensor([-1])]; bool var_856_keep_dims_0 = const()[name = string("op_856_keep_dims_0"), val = bool(true)]; tensor var_856 = reduce_sum(axes = var_856_axes_0, keep_dims = var_856_keep_dims_0, x = var_846)[name = string("op_856")]; tensor var_862_cast_fp16 = real_div(x = var_846, y = var_856)[name = string("op_862_cast_fp16")]; bool attn_output_1_transpose_x_0 = const()[name = string("attn_output_1_transpose_x_0"), val = bool(false)]; bool attn_output_1_transpose_y_0 = const()[name = string("attn_output_1_transpose_y_0"), val = bool(false)]; tensor V_expanded_1_cast_fp16 = transpose(perm = V_expanded_1_perm_0, x = reshape_3_cast_fp16)[name = string("transpose_71")]; tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_0, transpose_y = attn_output_1_transpose_y_0, x = var_862_cast_fp16, y = V_expanded_1_cast_fp16)[name = string("attn_output_1_cast_fp16")]; tensor var_873 = const()[name = string("op_873"), val = tensor([0, 2, 1, 3])]; tensor var_880 = const()[name = string("op_880"), val = tensor([1, 1, -1])]; tensor var_874_cast_fp16 = transpose(perm = var_873, x = attn_output_1_cast_fp16)[name = string("transpose_70")]; tensor input_17_cast_fp16 = reshape(shape = var_880, x = var_874_cast_fp16)[name = string("input_17_cast_fp16")]; tensor layers_0_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139607104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141180032))))[name = string("layers_0_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_4_bias_0_to_fp16 = const()[name = string("linear_4_bias_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141181632)))]; tensor linear_4_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_0_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_17_cast_fp16)[name = string("linear_4_cast_fp16")]; int32 var_889 = const()[name = string("op_889"), val = int32(-1)]; fp16 const_8_promoted_to_fp16 = const()[name = string("const_8_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_891_cast_fp16 = mul(x = linear_4_cast_fp16, y = const_8_promoted_to_fp16)[name = string("op_891_cast_fp16")]; bool input_19_interleave_0 = const()[name = string("input_19_interleave_0"), val = bool(false)]; tensor input_19_cast_fp16 = concat(axis = var_889, interleave = input_19_interleave_0, values = (linear_4_cast_fp16, var_891_cast_fp16))[name = string("input_19_cast_fp16")]; tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; fp16 var_886_to_fp16 = const()[name = string("op_886_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_886_to_fp16, x = input_19_cast_fp16)[name = string("normed_17_cast_fp16")]; tensor var_896_split_sizes_0 = const()[name = string("op_896_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_896_axis_0 = const()[name = string("op_896_axis_0"), val = int32(-1)]; tensor var_896_cast_fp16_0, tensor var_896_cast_fp16_1 = split(axis = var_896_axis_0, split_sizes = var_896_split_sizes_0, x = normed_17_cast_fp16)[name = string("op_896_cast_fp16")]; tensor layers_0_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_0_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141184768)))]; tensor attn_output_3_cast_fp16 = mul(x = var_896_cast_fp16_0, y = layers_0_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_3_cast_fp16")]; tensor x_13_cast_fp16 = add(x = hidden_states, y = attn_output_3_cast_fp16)[name = string("x_13_cast_fp16")]; int32 var_905 = const()[name = string("op_905"), val = int32(-1)]; fp16 const_9_promoted_to_fp16 = const()[name = string("const_9_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_907_cast_fp16 = mul(x = x_13_cast_fp16, y = const_9_promoted_to_fp16)[name = string("op_907_cast_fp16")]; bool input_21_interleave_0 = const()[name = string("input_21_interleave_0"), val = bool(false)]; tensor input_21_cast_fp16 = concat(axis = var_905, interleave = input_21_interleave_0, values = (x_13_cast_fp16, var_907_cast_fp16))[name = string("input_21_cast_fp16")]; tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; fp16 var_902_to_fp16 = const()[name = string("op_902_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_902_to_fp16, x = input_21_cast_fp16)[name = string("normed_21_cast_fp16")]; tensor var_912_split_sizes_0 = const()[name = string("op_912_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_912_axis_0 = const()[name = string("op_912_axis_0"), val = int32(-1)]; tensor var_912_cast_fp16_0, tensor var_912_cast_fp16_1 = split(axis = var_912_axis_0, split_sizes = var_912_split_sizes_0, x = normed_21_cast_fp16)[name = string("op_912_cast_fp16")]; tensor layers_0_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_0_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141187904)))]; tensor var_914_cast_fp16 = mul(x = var_912_cast_fp16_0, y = layers_0_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("op_914_cast_fp16")]; tensor linear_5_bias_0 = const()[name = string("linear_5_bias_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141191040)))]; tensor gate_1 = linear(bias = linear_5_bias_0, weight = layers_0_mlp_gate_proj_weight_palettized, x = var_914_cast_fp16)[name = string("linear_5")]; tensor up_1 = linear(bias = linear_5_bias_0, weight = layers_0_mlp_up_proj_weight_palettized, x = var_914_cast_fp16)[name = string("linear_6")]; string gate_3_mode_0 = const()[name = string("gate_3_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_3 = gelu(mode = gate_3_mode_0, x = gate_1)[name = string("gate_3")]; tensor input_25 = mul(x = gate_3, y = up_1)[name = string("input_25")]; tensor x_15 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_0_mlp_down_proj_weight_palettized, x = input_25)[name = string("linear_7")]; int32 var_936 = const()[name = string("op_936"), val = int32(-1)]; fp16 const_10_promoted = const()[name = string("const_10_promoted"), val = fp16(-0x1p+0)]; tensor var_938 = mul(x = x_15, y = const_10_promoted)[name = string("op_938")]; bool input_27_interleave_0 = const()[name = string("input_27_interleave_0"), val = bool(false)]; tensor input_27 = concat(axis = var_936, interleave = input_27_interleave_0, values = (x_15, var_938))[name = string("input_27")]; tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; fp16 var_933_to_fp16 = const()[name = string("op_933_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_933_to_fp16, x = input_27)[name = string("normed_25_cast_fp16")]; tensor var_943_split_sizes_0 = const()[name = string("op_943_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_943_axis_0 = const()[name = string("op_943_axis_0"), val = int32(-1)]; tensor var_943_0, tensor var_943_1 = split(axis = var_943_axis_0, split_sizes = var_943_split_sizes_0, x = normed_25_cast_fp16)[name = string("op_943")]; tensor hidden_states_3 = mul(x = var_943_0, y = layers_0_post_feedforward_layernorm_weight)[name = string("hidden_states_3")]; tensor hidden_states_5_cast_fp16 = add(x = x_13_cast_fp16, y = hidden_states_3)[name = string("hidden_states_5_cast_fp16")]; tensor per_layer_slice_1_begin_0 = const()[name = string("per_layer_slice_1_begin_0"), val = tensor([0, 0, 0])]; tensor per_layer_slice_1_end_0 = const()[name = string("per_layer_slice_1_end_0"), val = tensor([1, 1, 256])]; tensor per_layer_slice_1_end_mask_0 = const()[name = string("per_layer_slice_1_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_1_cast_fp16 = slice_by_index(begin = per_layer_slice_1_begin_0, end = per_layer_slice_1_end_0, end_mask = per_layer_slice_1_end_mask_0, x = per_layer_combined_out)[name = string("per_layer_slice_1_cast_fp16")]; tensor gated_1 = linear(bias = linear_2_bias_0, weight = layers_0_per_layer_input_gate_weight_palettized, x = hidden_states_5_cast_fp16)[name = string("linear_8")]; string gated_3_mode_0 = const()[name = string("gated_3_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_3 = gelu(mode = gated_3_mode_0, x = gated_1)[name = string("gated_3")]; tensor input_31_cast_fp16 = mul(x = gated_3, y = per_layer_slice_1_cast_fp16)[name = string("input_31_cast_fp16")]; tensor layers_0_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141203392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141400064))))[name = string("layers_0_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_9_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_0_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_31_cast_fp16)[name = string("linear_9_cast_fp16")]; int32 var_981 = const()[name = string("op_981"), val = int32(-1)]; fp16 const_11_promoted_to_fp16 = const()[name = string("const_11_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_983_cast_fp16 = mul(x = linear_9_cast_fp16, y = const_11_promoted_to_fp16)[name = string("op_983_cast_fp16")]; bool input_33_interleave_0 = const()[name = string("input_33_interleave_0"), val = bool(false)]; tensor input_33_cast_fp16 = concat(axis = var_981, interleave = input_33_interleave_0, values = (linear_9_cast_fp16, var_983_cast_fp16))[name = string("input_33_cast_fp16")]; tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; fp16 var_978_to_fp16 = const()[name = string("op_978_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_978_to_fp16, x = input_33_cast_fp16)[name = string("normed_29_cast_fp16")]; tensor var_988_split_sizes_0 = const()[name = string("op_988_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_988_axis_0 = const()[name = string("op_988_axis_0"), val = int32(-1)]; tensor var_988_cast_fp16_0, tensor var_988_cast_fp16_1 = split(axis = var_988_axis_0, split_sizes = var_988_split_sizes_0, x = normed_29_cast_fp16)[name = string("op_988_cast_fp16")]; tensor layers_0_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_0_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141401664)))]; tensor hidden_states_7_cast_fp16 = mul(x = var_988_cast_fp16_0, y = layers_0_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_7_cast_fp16")]; tensor hidden_states_9_cast_fp16 = add(x = hidden_states_5_cast_fp16, y = hidden_states_7_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; tensor const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = tensor([0x1.24p-6])]; tensor x_19_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_12_promoted_to_fp16)[name = string("x_19_cast_fp16")]; int32 var_1003 = const()[name = string("op_1003"), val = int32(-1)]; fp16 const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1005_cast_fp16 = mul(x = x_19_cast_fp16, y = const_13_promoted_to_fp16)[name = string("op_1005_cast_fp16")]; bool input_35_interleave_0 = const()[name = string("input_35_interleave_0"), val = bool(false)]; tensor input_35_cast_fp16 = concat(axis = var_1003, interleave = input_35_interleave_0, values = (x_19_cast_fp16, var_1005_cast_fp16))[name = string("input_35_cast_fp16")]; tensor normed_33_axes_0 = const()[name = string("normed_33_axes_0"), val = tensor([-1])]; fp16 var_1000_to_fp16 = const()[name = string("op_1000_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_33_cast_fp16 = layer_norm(axes = normed_33_axes_0, epsilon = var_1000_to_fp16, x = input_35_cast_fp16)[name = string("normed_33_cast_fp16")]; tensor var_1010_split_sizes_0 = const()[name = string("op_1010_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1010_axis_0 = const()[name = string("op_1010_axis_0"), val = int32(-1)]; tensor var_1010_cast_fp16_0, tensor var_1010_cast_fp16_1 = split(axis = var_1010_axis_0, split_sizes = var_1010_split_sizes_0, x = normed_33_cast_fp16)[name = string("op_1010_cast_fp16")]; tensor layers_1_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_1_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141404800)))]; tensor var_1012_cast_fp16 = mul(x = var_1010_cast_fp16_0, y = layers_1_input_layernorm_weight_promoted_to_fp16)[name = string("op_1012_cast_fp16")]; tensor var_1020 = linear(bias = linear_1_bias_0, weight = layers_1_self_attn_q_proj_weight_palettized, x = var_1012_cast_fp16)[name = string("linear_10")]; tensor var_1025 = const()[name = string("op_1025"), val = tensor([1, 8, 256, 1])]; tensor var_1026 = reshape(shape = var_1025, x = var_1020)[name = string("op_1026")]; tensor var_1031 = const()[name = string("op_1031"), val = tensor([0, 1, 3, 2])]; tensor var_1041 = const()[name = string("op_1041"), val = tensor([1, 8, 256])]; tensor var_1032 = transpose(perm = var_1031, x = var_1026)[name = string("transpose_69")]; tensor x_21 = reshape(shape = var_1041, x = var_1032)[name = string("x_21")]; int32 var_1047 = const()[name = string("op_1047"), val = int32(-1)]; fp16 const_14_promoted = const()[name = string("const_14_promoted"), val = fp16(-0x1p+0)]; tensor var_1049 = mul(x = x_21, y = const_14_promoted)[name = string("op_1049")]; bool input_39_interleave_0 = const()[name = string("input_39_interleave_0"), val = bool(false)]; tensor input_39 = concat(axis = var_1047, interleave = input_39_interleave_0, values = (x_21, var_1049))[name = string("input_39")]; tensor normed_37_axes_0 = const()[name = string("normed_37_axes_0"), val = tensor([-1])]; fp16 var_1044_to_fp16 = const()[name = string("op_1044_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_37_cast_fp16 = layer_norm(axes = normed_37_axes_0, epsilon = var_1044_to_fp16, x = input_39)[name = string("normed_37_cast_fp16")]; tensor var_1054_split_sizes_0 = const()[name = string("op_1054_split_sizes_0"), val = tensor([256, 256])]; int32 var_1054_axis_0 = const()[name = string("op_1054_axis_0"), val = int32(-1)]; tensor var_1054_0, tensor var_1054_1 = split(axis = var_1054_axis_0, split_sizes = var_1054_split_sizes_0, x = normed_37_cast_fp16)[name = string("op_1054")]; tensor var_1056 = mul(x = var_1054_0, y = layers_1_self_attn_q_norm_weight)[name = string("op_1056")]; tensor var_1061 = const()[name = string("op_1061"), val = tensor([1, 8, 1, 256])]; tensor q_11 = reshape(shape = var_1061, x = var_1056)[name = string("q_11")]; tensor var_1063_cast_fp16 = mul(x = q_11, y = cos_s)[name = string("op_1063_cast_fp16")]; tensor var_1064_split_sizes_0 = const()[name = string("op_1064_split_sizes_0"), val = tensor([128, 128])]; int32 var_1064_axis_0 = const()[name = string("op_1064_axis_0"), val = int32(-1)]; tensor var_1064_0, tensor var_1064_1 = split(axis = var_1064_axis_0, split_sizes = var_1064_split_sizes_0, x = q_11)[name = string("op_1064")]; fp16 const_15_promoted = const()[name = string("const_15_promoted"), val = fp16(-0x1p+0)]; tensor var_1066 = mul(x = var_1064_1, y = const_15_promoted)[name = string("op_1066")]; int32 var_1068 = const()[name = string("op_1068"), val = int32(-1)]; bool var_1069_interleave_0 = const()[name = string("op_1069_interleave_0"), val = bool(false)]; tensor var_1069 = concat(axis = var_1068, interleave = var_1069_interleave_0, values = (var_1066, var_1064_0))[name = string("op_1069")]; tensor var_1070_cast_fp16 = mul(x = var_1069, y = sin_s)[name = string("op_1070_cast_fp16")]; tensor q_15_cast_fp16 = add(x = var_1063_cast_fp16, y = var_1070_cast_fp16)[name = string("q_15_cast_fp16")]; tensor var_1075 = linear(bias = linear_2_bias_0, weight = layers_1_self_attn_k_proj_weight_palettized, x = var_1012_cast_fp16)[name = string("linear_11")]; tensor var_1080 = const()[name = string("op_1080"), val = tensor([1, 1, 256, 1])]; tensor var_1081 = reshape(shape = var_1080, x = var_1075)[name = string("op_1081")]; tensor var_1086 = const()[name = string("op_1086"), val = tensor([0, 1, 3, 2])]; tensor var_1095 = linear(bias = linear_2_bias_0, weight = layers_1_self_attn_v_proj_weight_palettized, x = var_1012_cast_fp16)[name = string("linear_12")]; tensor var_1100 = const()[name = string("op_1100"), val = tensor([1, 1, 256, 1])]; tensor var_1101 = reshape(shape = var_1100, x = var_1095)[name = string("op_1101")]; tensor var_1106 = const()[name = string("op_1106"), val = tensor([0, 1, 3, 2])]; tensor var_1116 = const()[name = string("op_1116"), val = tensor([1, 1, 256])]; tensor var_1087 = transpose(perm = var_1086, x = var_1081)[name = string("transpose_68")]; tensor x_23 = reshape(shape = var_1116, x = var_1087)[name = string("x_23")]; int32 var_1122 = const()[name = string("op_1122"), val = int32(-1)]; fp16 const_16_promoted = const()[name = string("const_16_promoted"), val = fp16(-0x1p+0)]; tensor var_1124 = mul(x = x_23, y = const_16_promoted)[name = string("op_1124")]; bool input_41_interleave_0 = const()[name = string("input_41_interleave_0"), val = bool(false)]; tensor input_41 = concat(axis = var_1122, interleave = input_41_interleave_0, values = (x_23, var_1124))[name = string("input_41")]; tensor normed_41_axes_0 = const()[name = string("normed_41_axes_0"), val = tensor([-1])]; fp16 var_1119_to_fp16 = const()[name = string("op_1119_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_41_cast_fp16 = layer_norm(axes = normed_41_axes_0, epsilon = var_1119_to_fp16, x = input_41)[name = string("normed_41_cast_fp16")]; tensor var_1129_split_sizes_0 = const()[name = string("op_1129_split_sizes_0"), val = tensor([256, 256])]; int32 var_1129_axis_0 = const()[name = string("op_1129_axis_0"), val = int32(-1)]; tensor var_1129_0, tensor var_1129_1 = split(axis = var_1129_axis_0, split_sizes = var_1129_split_sizes_0, x = normed_41_cast_fp16)[name = string("op_1129")]; tensor var_1131 = mul(x = var_1129_0, y = layers_1_self_attn_k_norm_weight)[name = string("op_1131")]; tensor var_1136 = const()[name = string("op_1136"), val = tensor([1, 1, 1, 256])]; tensor q_13 = reshape(shape = var_1136, x = var_1131)[name = string("q_13")]; fp16 var_1138_promoted = const()[name = string("op_1138_promoted"), val = fp16(0x1p+1)]; tensor var_1107 = transpose(perm = var_1106, x = var_1101)[name = string("transpose_67")]; tensor var_1139 = pow(x = var_1107, y = var_1138_promoted)[name = string("op_1139")]; tensor var_1144_axes_0 = const()[name = string("op_1144_axes_0"), val = tensor([-1])]; bool var_1144_keep_dims_0 = const()[name = string("op_1144_keep_dims_0"), val = bool(true)]; tensor var_1144 = reduce_mean(axes = var_1144_axes_0, keep_dims = var_1144_keep_dims_0, x = var_1139)[name = string("op_1144")]; fp16 var_1146_to_fp16 = const()[name = string("op_1146_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_3_cast_fp16 = add(x = var_1144, y = var_1146_to_fp16)[name = string("mean_sq_3_cast_fp16")]; fp32 var_1148_epsilon_0 = const()[name = string("op_1148_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_1148_cast_fp16 = rsqrt(epsilon = var_1148_epsilon_0, x = mean_sq_3_cast_fp16)[name = string("op_1148_cast_fp16")]; tensor input_45_cast_fp16 = mul(x = var_1107, y = var_1148_cast_fp16)[name = string("input_45_cast_fp16")]; tensor var_1150_cast_fp16 = mul(x = q_13, y = cos_s)[name = string("op_1150_cast_fp16")]; tensor var_1151_split_sizes_0 = const()[name = string("op_1151_split_sizes_0"), val = tensor([128, 128])]; int32 var_1151_axis_0 = const()[name = string("op_1151_axis_0"), val = int32(-1)]; tensor var_1151_0, tensor var_1151_1 = split(axis = var_1151_axis_0, split_sizes = var_1151_split_sizes_0, x = q_13)[name = string("op_1151")]; fp16 const_17_promoted = const()[name = string("const_17_promoted"), val = fp16(-0x1p+0)]; tensor var_1153 = mul(x = var_1151_1, y = const_17_promoted)[name = string("op_1153")]; int32 var_1155 = const()[name = string("op_1155"), val = int32(-1)]; bool var_1156_interleave_0 = const()[name = string("op_1156_interleave_0"), val = bool(false)]; tensor var_1156 = concat(axis = var_1155, interleave = var_1156_interleave_0, values = (var_1153, var_1151_0))[name = string("op_1156")]; tensor var_1157_cast_fp16 = mul(x = var_1156, y = sin_s)[name = string("op_1157_cast_fp16")]; tensor input_43_cast_fp16 = add(x = var_1150_cast_fp16, y = var_1157_cast_fp16)[name = string("input_43_cast_fp16")]; tensor k_padded_3_pad_0 = const()[name = string("k_padded_3_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_3_mode_0 = const()[name = string("k_padded_3_mode_0"), val = string("constant")]; fp16 const_18_to_fp16 = const()[name = string("const_18_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_3_cast_fp16 = pad(constant_val = const_18_to_fp16, mode = k_padded_3_mode_0, pad = k_padded_3_pad_0, x = input_43_cast_fp16)[name = string("k_padded_3_cast_fp16")]; tensor v_padded_3_pad_0 = const()[name = string("v_padded_3_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_3_mode_0 = const()[name = string("v_padded_3_mode_0"), val = string("constant")]; fp16 const_19_to_fp16 = const()[name = string("const_19_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_3_cast_fp16 = pad(constant_val = const_19_to_fp16, mode = v_padded_3_mode_0, pad = v_padded_3_pad_0, x = input_45_cast_fp16)[name = string("v_padded_3_cast_fp16")]; tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([2])]; tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([3])]; int32 concat_14_axis_0 = const()[name = string("concat_14_axis_0"), val = int32(0)]; bool concat_14_interleave_0 = const()[name = string("concat_14_interleave_0"), val = bool(false)]; tensor concat_14 = concat(axis = concat_14_axis_0, interleave = concat_14_interleave_0, values = (expand_dims_12, expand_dims_13, ring_pos, expand_dims_15))[name = string("concat_14")]; tensor concat_15_values1_0 = const()[name = string("concat_15_values1_0"), val = tensor([0])]; tensor concat_15_values3_0 = const()[name = string("concat_15_values3_0"), val = tensor([0])]; int32 concat_15_axis_0 = const()[name = string("concat_15_axis_0"), val = int32(0)]; bool concat_15_interleave_0 = const()[name = string("concat_15_interleave_0"), val = bool(false)]; tensor concat_15 = concat(axis = concat_15_axis_0, interleave = concat_15_interleave_0, values = (expand_dims_16, concat_15_values1_0, var_725, concat_15_values3_0))[name = string("concat_15")]; tensor kv_cache_sliding_internal_tensor_assign_3_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_3_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_3_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_14, begin_mask = kv_cache_sliding_internal_tensor_assign_3_begin_mask_0, end = concat_15, end_mask = kv_cache_sliding_internal_tensor_assign_3_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_3_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_3_stride_0, update = k_padded_3_cast_fp16, x = coreml_update_state_17)[name = string("kv_cache_sliding_internal_tensor_assign_3_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_3_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_2_write_state")]; tensor coreml_update_state_18 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_2")]; tensor expand_dims_18 = const()[name = string("expand_dims_18"), val = tensor([3])]; tensor expand_dims_19 = const()[name = string("expand_dims_19"), val = tensor([0])]; tensor expand_dims_21 = const()[name = string("expand_dims_21"), val = tensor([0])]; tensor expand_dims_22 = const()[name = string("expand_dims_22"), val = tensor([4])]; int32 concat_18_axis_0 = const()[name = string("concat_18_axis_0"), val = int32(0)]; bool concat_18_interleave_0 = const()[name = string("concat_18_interleave_0"), val = bool(false)]; tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (expand_dims_18, expand_dims_19, ring_pos, expand_dims_21))[name = string("concat_18")]; tensor concat_19_values1_0 = const()[name = string("concat_19_values1_0"), val = tensor([0])]; tensor concat_19_values3_0 = const()[name = string("concat_19_values3_0"), val = tensor([0])]; int32 concat_19_axis_0 = const()[name = string("concat_19_axis_0"), val = int32(0)]; bool concat_19_interleave_0 = const()[name = string("concat_19_interleave_0"), val = bool(false)]; tensor concat_19 = concat(axis = concat_19_axis_0, interleave = concat_19_interleave_0, values = (expand_dims_22, concat_19_values1_0, var_725, concat_19_values3_0))[name = string("concat_19")]; tensor kv_cache_sliding_internal_tensor_assign_4_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_4_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_4_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_18, begin_mask = kv_cache_sliding_internal_tensor_assign_4_begin_mask_0, end = concat_19, end_mask = kv_cache_sliding_internal_tensor_assign_4_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_4_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_4_stride_0, update = v_padded_3_cast_fp16, x = coreml_update_state_18)[name = string("kv_cache_sliding_internal_tensor_assign_4_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_4_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_3_write_state")]; tensor coreml_update_state_19 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_3")]; tensor var_1224_begin_0 = const()[name = string("op_1224_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1224_end_0 = const()[name = string("op_1224_end_0"), val = tensor([3, 1, 512, 512])]; tensor var_1224_end_mask_0 = const()[name = string("op_1224_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1224_cast_fp16 = slice_by_index(begin = var_1224_begin_0, end = var_1224_end_0, end_mask = var_1224_end_mask_0, x = coreml_update_state_19)[name = string("op_1224_cast_fp16")]; tensor K_sliding_slice_3_begin_0 = const()[name = string("K_sliding_slice_3_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_sliding_slice_3_end_0 = const()[name = string("K_sliding_slice_3_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_sliding_slice_3_end_mask_0 = const()[name = string("K_sliding_slice_3_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_sliding_slice_3_cast_fp16 = slice_by_index(begin = K_sliding_slice_3_begin_0, end = K_sliding_slice_3_end_0, end_mask = K_sliding_slice_3_end_mask_0, x = var_1224_cast_fp16)[name = string("K_sliding_slice_3_cast_fp16")]; tensor var_1244_begin_0 = const()[name = string("op_1244_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_1244_end_0 = const()[name = string("op_1244_end_0"), val = tensor([4, 1, 512, 512])]; tensor var_1244_end_mask_0 = const()[name = string("op_1244_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1244_cast_fp16 = slice_by_index(begin = var_1244_begin_0, end = var_1244_end_0, end_mask = var_1244_end_mask_0, x = coreml_update_state_19)[name = string("op_1244_cast_fp16")]; tensor V_for_attn_3_begin_0 = const()[name = string("V_for_attn_3_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_3_end_0 = const()[name = string("V_for_attn_3_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_3_end_mask_0 = const()[name = string("V_for_attn_3_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_3_cast_fp16 = slice_by_index(begin = V_for_attn_3_begin_0, end = V_for_attn_3_end_0, end_mask = V_for_attn_3_end_mask_0, x = var_1244_cast_fp16)[name = string("V_for_attn_3_cast_fp16")]; tensor transpose_4_perm_0 = const()[name = string("transpose_4_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_2_reps_0 = const()[name = string("tile_2_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_4_cast_fp16 = transpose(perm = transpose_4_perm_0, x = K_sliding_slice_3_cast_fp16)[name = string("transpose_66")]; tensor tile_2_cast_fp16 = tile(reps = tile_2_reps_0, x = transpose_4_cast_fp16)[name = string("tile_2_cast_fp16")]; tensor concat_20 = const()[name = string("concat_20"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_4_cast_fp16 = reshape(shape = concat_20, x = tile_2_cast_fp16)[name = string("reshape_4_cast_fp16")]; tensor transpose_5_perm_0 = const()[name = string("transpose_5_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_21 = const()[name = string("concat_21"), val = tensor([-1, 1, 512, 256])]; tensor transpose_5_cast_fp16 = transpose(perm = transpose_5_perm_0, x = reshape_4_cast_fp16)[name = string("transpose_65")]; tensor reshape_5_cast_fp16 = reshape(shape = concat_21, x = transpose_5_cast_fp16)[name = string("reshape_5_cast_fp16")]; tensor transpose_33_perm_0 = const()[name = string("transpose_33_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_6_perm_0 = const()[name = string("transpose_6_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_3_reps_0 = const()[name = string("tile_3_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_6_cast_fp16 = transpose(perm = transpose_6_perm_0, x = V_for_attn_3_cast_fp16)[name = string("transpose_64")]; tensor tile_3_cast_fp16 = tile(reps = tile_3_reps_0, x = transpose_6_cast_fp16)[name = string("tile_3_cast_fp16")]; tensor concat_22 = const()[name = string("concat_22"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_6_cast_fp16 = reshape(shape = concat_22, x = tile_3_cast_fp16)[name = string("reshape_6_cast_fp16")]; tensor transpose_7_perm_0 = const()[name = string("transpose_7_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_23 = const()[name = string("concat_23"), val = tensor([-1, 1, 512, 256])]; tensor transpose_7_cast_fp16 = transpose(perm = transpose_7_perm_0, x = reshape_6_cast_fp16)[name = string("transpose_63")]; tensor reshape_7_cast_fp16 = reshape(shape = concat_23, x = transpose_7_cast_fp16)[name = string("reshape_7_cast_fp16")]; tensor V_expanded_3_perm_0 = const()[name = string("V_expanded_3_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_5_transpose_x_0 = const()[name = string("attn_weights_5_transpose_x_0"), val = bool(false)]; bool attn_weights_5_transpose_y_0 = const()[name = string("attn_weights_5_transpose_y_0"), val = bool(false)]; tensor transpose_33_cast_fp16 = transpose(perm = transpose_33_perm_0, x = reshape_5_cast_fp16)[name = string("transpose_62")]; tensor attn_weights_5_cast_fp16 = matmul(transpose_x = attn_weights_5_transpose_x_0, transpose_y = attn_weights_5_transpose_y_0, x = q_15_cast_fp16, y = transpose_33_cast_fp16)[name = string("attn_weights_5_cast_fp16")]; tensor x_27_cast_fp16 = add(x = attn_weights_5_cast_fp16, y = causal_mask_sliding)[name = string("x_27_cast_fp16")]; tensor reduce_max_1_axes_0 = const()[name = string("reduce_max_1_axes_0"), val = tensor([-1])]; bool reduce_max_1_keep_dims_0 = const()[name = string("reduce_max_1_keep_dims_0"), val = bool(true)]; tensor reduce_max_1 = reduce_max(axes = reduce_max_1_axes_0, keep_dims = reduce_max_1_keep_dims_0, x = x_27_cast_fp16)[name = string("reduce_max_1")]; tensor var_1289 = sub(x = x_27_cast_fp16, y = reduce_max_1)[name = string("op_1289")]; tensor var_1295 = exp(x = var_1289)[name = string("op_1295")]; tensor var_1305_axes_0 = const()[name = string("op_1305_axes_0"), val = tensor([-1])]; bool var_1305_keep_dims_0 = const()[name = string("op_1305_keep_dims_0"), val = bool(true)]; tensor var_1305 = reduce_sum(axes = var_1305_axes_0, keep_dims = var_1305_keep_dims_0, x = var_1295)[name = string("op_1305")]; tensor var_1311_cast_fp16 = real_div(x = var_1295, y = var_1305)[name = string("op_1311_cast_fp16")]; bool attn_output_5_transpose_x_0 = const()[name = string("attn_output_5_transpose_x_0"), val = bool(false)]; bool attn_output_5_transpose_y_0 = const()[name = string("attn_output_5_transpose_y_0"), val = bool(false)]; tensor V_expanded_3_cast_fp16 = transpose(perm = V_expanded_3_perm_0, x = reshape_7_cast_fp16)[name = string("transpose_61")]; tensor attn_output_5_cast_fp16 = matmul(transpose_x = attn_output_5_transpose_x_0, transpose_y = attn_output_5_transpose_y_0, x = var_1311_cast_fp16, y = V_expanded_3_cast_fp16)[name = string("attn_output_5_cast_fp16")]; tensor var_1322 = const()[name = string("op_1322"), val = tensor([0, 2, 1, 3])]; tensor var_1329 = const()[name = string("op_1329"), val = tensor([1, 1, -1])]; tensor var_1323_cast_fp16 = transpose(perm = var_1322, x = attn_output_5_cast_fp16)[name = string("transpose_60")]; tensor input_47_cast_fp16 = reshape(shape = var_1329, x = var_1323_cast_fp16)[name = string("input_47_cast_fp16")]; tensor layers_1_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141407936))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(142980864))))[name = string("layers_1_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_13_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_1_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_47_cast_fp16)[name = string("linear_13_cast_fp16")]; int32 var_1338 = const()[name = string("op_1338"), val = int32(-1)]; fp16 const_20_promoted_to_fp16 = const()[name = string("const_20_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1340_cast_fp16 = mul(x = linear_13_cast_fp16, y = const_20_promoted_to_fp16)[name = string("op_1340_cast_fp16")]; bool input_49_interleave_0 = const()[name = string("input_49_interleave_0"), val = bool(false)]; tensor input_49_cast_fp16 = concat(axis = var_1338, interleave = input_49_interleave_0, values = (linear_13_cast_fp16, var_1340_cast_fp16))[name = string("input_49_cast_fp16")]; tensor normed_45_axes_0 = const()[name = string("normed_45_axes_0"), val = tensor([-1])]; fp16 var_1335_to_fp16 = const()[name = string("op_1335_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_45_cast_fp16 = layer_norm(axes = normed_45_axes_0, epsilon = var_1335_to_fp16, x = input_49_cast_fp16)[name = string("normed_45_cast_fp16")]; tensor var_1345_split_sizes_0 = const()[name = string("op_1345_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1345_axis_0 = const()[name = string("op_1345_axis_0"), val = int32(-1)]; tensor var_1345_cast_fp16_0, tensor var_1345_cast_fp16_1 = split(axis = var_1345_axis_0, split_sizes = var_1345_split_sizes_0, x = normed_45_cast_fp16)[name = string("op_1345_cast_fp16")]; tensor layers_1_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_1_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(142982464)))]; tensor attn_output_7_cast_fp16 = mul(x = var_1345_cast_fp16_0, y = layers_1_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_7_cast_fp16")]; tensor x_33_cast_fp16 = add(x = x_19_cast_fp16, y = attn_output_7_cast_fp16)[name = string("x_33_cast_fp16")]; int32 var_1354 = const()[name = string("op_1354"), val = int32(-1)]; fp16 const_21_promoted_to_fp16 = const()[name = string("const_21_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1356_cast_fp16 = mul(x = x_33_cast_fp16, y = const_21_promoted_to_fp16)[name = string("op_1356_cast_fp16")]; bool input_51_interleave_0 = const()[name = string("input_51_interleave_0"), val = bool(false)]; tensor input_51_cast_fp16 = concat(axis = var_1354, interleave = input_51_interleave_0, values = (x_33_cast_fp16, var_1356_cast_fp16))[name = string("input_51_cast_fp16")]; tensor normed_49_axes_0 = const()[name = string("normed_49_axes_0"), val = tensor([-1])]; fp16 var_1351_to_fp16 = const()[name = string("op_1351_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_49_cast_fp16 = layer_norm(axes = normed_49_axes_0, epsilon = var_1351_to_fp16, x = input_51_cast_fp16)[name = string("normed_49_cast_fp16")]; tensor var_1361_split_sizes_0 = const()[name = string("op_1361_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1361_axis_0 = const()[name = string("op_1361_axis_0"), val = int32(-1)]; tensor var_1361_cast_fp16_0, tensor var_1361_cast_fp16_1 = split(axis = var_1361_axis_0, split_sizes = var_1361_split_sizes_0, x = normed_49_cast_fp16)[name = string("op_1361_cast_fp16")]; tensor layers_1_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_1_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(142985600)))]; tensor var_1363_cast_fp16 = mul(x = var_1361_cast_fp16_0, y = layers_1_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("op_1363_cast_fp16")]; tensor gate_5 = linear(bias = linear_5_bias_0, weight = layers_1_mlp_gate_proj_weight_palettized, x = var_1363_cast_fp16)[name = string("linear_14")]; tensor up_3 = linear(bias = linear_5_bias_0, weight = layers_1_mlp_up_proj_weight_palettized, x = var_1363_cast_fp16)[name = string("linear_15")]; string gate_7_mode_0 = const()[name = string("gate_7_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_7 = gelu(mode = gate_7_mode_0, x = gate_5)[name = string("gate_7")]; tensor input_55 = mul(x = gate_7, y = up_3)[name = string("input_55")]; tensor x_35 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_1_mlp_down_proj_weight_palettized, x = input_55)[name = string("linear_16")]; int32 var_1385 = const()[name = string("op_1385"), val = int32(-1)]; fp16 const_22_promoted = const()[name = string("const_22_promoted"), val = fp16(-0x1p+0)]; tensor var_1387 = mul(x = x_35, y = const_22_promoted)[name = string("op_1387")]; bool input_57_interleave_0 = const()[name = string("input_57_interleave_0"), val = bool(false)]; tensor input_57 = concat(axis = var_1385, interleave = input_57_interleave_0, values = (x_35, var_1387))[name = string("input_57")]; tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; fp16 var_1382_to_fp16 = const()[name = string("op_1382_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_1382_to_fp16, x = input_57)[name = string("normed_53_cast_fp16")]; tensor var_1392_split_sizes_0 = const()[name = string("op_1392_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1392_axis_0 = const()[name = string("op_1392_axis_0"), val = int32(-1)]; tensor var_1392_0, tensor var_1392_1 = split(axis = var_1392_axis_0, split_sizes = var_1392_split_sizes_0, x = normed_53_cast_fp16)[name = string("op_1392")]; tensor hidden_states_11 = mul(x = var_1392_0, y = layers_1_post_feedforward_layernorm_weight)[name = string("hidden_states_11")]; tensor hidden_states_13_cast_fp16 = add(x = x_33_cast_fp16, y = hidden_states_11)[name = string("hidden_states_13_cast_fp16")]; tensor per_layer_slice_3_begin_0 = const()[name = string("per_layer_slice_3_begin_0"), val = tensor([0, 0, 256])]; tensor per_layer_slice_3_end_0 = const()[name = string("per_layer_slice_3_end_0"), val = tensor([1, 1, 512])]; tensor per_layer_slice_3_end_mask_0 = const()[name = string("per_layer_slice_3_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_3_cast_fp16 = slice_by_index(begin = per_layer_slice_3_begin_0, end = per_layer_slice_3_end_0, end_mask = per_layer_slice_3_end_mask_0, x = per_layer_combined_out)[name = string("per_layer_slice_3_cast_fp16")]; tensor gated_5 = linear(bias = linear_2_bias_0, weight = layers_1_per_layer_input_gate_weight_palettized, x = hidden_states_13_cast_fp16)[name = string("linear_17")]; string gated_7_mode_0 = const()[name = string("gated_7_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_7 = gelu(mode = gated_7_mode_0, x = gated_5)[name = string("gated_7")]; tensor input_61_cast_fp16 = mul(x = gated_7, y = per_layer_slice_3_cast_fp16)[name = string("input_61_cast_fp16")]; tensor layers_1_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(142988736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(143185408))))[name = string("layers_1_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_18_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_1_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_61_cast_fp16)[name = string("linear_18_cast_fp16")]; int32 var_1430 = const()[name = string("op_1430"), val = int32(-1)]; fp16 const_23_promoted_to_fp16 = const()[name = string("const_23_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1432_cast_fp16 = mul(x = linear_18_cast_fp16, y = const_23_promoted_to_fp16)[name = string("op_1432_cast_fp16")]; bool input_63_interleave_0 = const()[name = string("input_63_interleave_0"), val = bool(false)]; tensor input_63_cast_fp16 = concat(axis = var_1430, interleave = input_63_interleave_0, values = (linear_18_cast_fp16, var_1432_cast_fp16))[name = string("input_63_cast_fp16")]; tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; fp16 var_1427_to_fp16 = const()[name = string("op_1427_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_1427_to_fp16, x = input_63_cast_fp16)[name = string("normed_57_cast_fp16")]; tensor var_1437_split_sizes_0 = const()[name = string("op_1437_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1437_axis_0 = const()[name = string("op_1437_axis_0"), val = int32(-1)]; tensor var_1437_cast_fp16_0, tensor var_1437_cast_fp16_1 = split(axis = var_1437_axis_0, split_sizes = var_1437_split_sizes_0, x = normed_57_cast_fp16)[name = string("op_1437_cast_fp16")]; tensor layers_1_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_1_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(143187008)))]; tensor hidden_states_15_cast_fp16 = mul(x = var_1437_cast_fp16_0, y = layers_1_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_15_cast_fp16")]; tensor hidden_states_17_cast_fp16 = add(x = hidden_states_13_cast_fp16, y = hidden_states_15_cast_fp16)[name = string("hidden_states_17_cast_fp16")]; tensor const_24_promoted_to_fp16 = const()[name = string("const_24_promoted_to_fp16"), val = tensor([0x1.c8p-3])]; tensor x_39_cast_fp16 = mul(x = hidden_states_17_cast_fp16, y = const_24_promoted_to_fp16)[name = string("x_39_cast_fp16")]; int32 var_1452 = const()[name = string("op_1452"), val = int32(-1)]; fp16 const_25_promoted_to_fp16 = const()[name = string("const_25_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1454_cast_fp16 = mul(x = x_39_cast_fp16, y = const_25_promoted_to_fp16)[name = string("op_1454_cast_fp16")]; bool input_65_interleave_0 = const()[name = string("input_65_interleave_0"), val = bool(false)]; tensor input_65_cast_fp16 = concat(axis = var_1452, interleave = input_65_interleave_0, values = (x_39_cast_fp16, var_1454_cast_fp16))[name = string("input_65_cast_fp16")]; tensor normed_61_axes_0 = const()[name = string("normed_61_axes_0"), val = tensor([-1])]; fp16 var_1449_to_fp16 = const()[name = string("op_1449_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_61_cast_fp16 = layer_norm(axes = normed_61_axes_0, epsilon = var_1449_to_fp16, x = input_65_cast_fp16)[name = string("normed_61_cast_fp16")]; tensor var_1459_split_sizes_0 = const()[name = string("op_1459_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1459_axis_0 = const()[name = string("op_1459_axis_0"), val = int32(-1)]; tensor var_1459_cast_fp16_0, tensor var_1459_cast_fp16_1 = split(axis = var_1459_axis_0, split_sizes = var_1459_split_sizes_0, x = normed_61_cast_fp16)[name = string("op_1459_cast_fp16")]; tensor layers_2_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_2_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(143190144)))]; tensor var_1461_cast_fp16 = mul(x = var_1459_cast_fp16_0, y = layers_2_input_layernorm_weight_promoted_to_fp16)[name = string("op_1461_cast_fp16")]; tensor var_1469 = linear(bias = linear_1_bias_0, weight = layers_2_self_attn_q_proj_weight_palettized, x = var_1461_cast_fp16)[name = string("linear_19")]; tensor var_1474 = const()[name = string("op_1474"), val = tensor([1, 8, 256, 1])]; tensor var_1475 = reshape(shape = var_1474, x = var_1469)[name = string("op_1475")]; tensor var_1480 = const()[name = string("op_1480"), val = tensor([0, 1, 3, 2])]; tensor var_1490 = const()[name = string("op_1490"), val = tensor([1, 8, 256])]; tensor var_1481 = transpose(perm = var_1480, x = var_1475)[name = string("transpose_59")]; tensor x_41 = reshape(shape = var_1490, x = var_1481)[name = string("x_41")]; int32 var_1496 = const()[name = string("op_1496"), val = int32(-1)]; fp16 const_26_promoted = const()[name = string("const_26_promoted"), val = fp16(-0x1p+0)]; tensor var_1498 = mul(x = x_41, y = const_26_promoted)[name = string("op_1498")]; bool input_69_interleave_0 = const()[name = string("input_69_interleave_0"), val = bool(false)]; tensor input_69 = concat(axis = var_1496, interleave = input_69_interleave_0, values = (x_41, var_1498))[name = string("input_69")]; tensor normed_65_axes_0 = const()[name = string("normed_65_axes_0"), val = tensor([-1])]; fp16 var_1493_to_fp16 = const()[name = string("op_1493_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_65_cast_fp16 = layer_norm(axes = normed_65_axes_0, epsilon = var_1493_to_fp16, x = input_69)[name = string("normed_65_cast_fp16")]; tensor var_1503_split_sizes_0 = const()[name = string("op_1503_split_sizes_0"), val = tensor([256, 256])]; int32 var_1503_axis_0 = const()[name = string("op_1503_axis_0"), val = int32(-1)]; tensor var_1503_0, tensor var_1503_1 = split(axis = var_1503_axis_0, split_sizes = var_1503_split_sizes_0, x = normed_65_cast_fp16)[name = string("op_1503")]; tensor var_1505 = mul(x = var_1503_0, y = layers_2_self_attn_q_norm_weight)[name = string("op_1505")]; tensor var_1510 = const()[name = string("op_1510"), val = tensor([1, 8, 1, 256])]; tensor q_19 = reshape(shape = var_1510, x = var_1505)[name = string("q_19")]; tensor var_1512_cast_fp16 = mul(x = q_19, y = cos_s)[name = string("op_1512_cast_fp16")]; tensor var_1513_split_sizes_0 = const()[name = string("op_1513_split_sizes_0"), val = tensor([128, 128])]; int32 var_1513_axis_0 = const()[name = string("op_1513_axis_0"), val = int32(-1)]; tensor var_1513_0, tensor var_1513_1 = split(axis = var_1513_axis_0, split_sizes = var_1513_split_sizes_0, x = q_19)[name = string("op_1513")]; fp16 const_27_promoted = const()[name = string("const_27_promoted"), val = fp16(-0x1p+0)]; tensor var_1515 = mul(x = var_1513_1, y = const_27_promoted)[name = string("op_1515")]; int32 var_1517 = const()[name = string("op_1517"), val = int32(-1)]; bool var_1518_interleave_0 = const()[name = string("op_1518_interleave_0"), val = bool(false)]; tensor var_1518 = concat(axis = var_1517, interleave = var_1518_interleave_0, values = (var_1515, var_1513_0))[name = string("op_1518")]; tensor var_1519_cast_fp16 = mul(x = var_1518, y = sin_s)[name = string("op_1519_cast_fp16")]; tensor q_23_cast_fp16 = add(x = var_1512_cast_fp16, y = var_1519_cast_fp16)[name = string("q_23_cast_fp16")]; tensor var_1524 = linear(bias = linear_2_bias_0, weight = layers_2_self_attn_k_proj_weight_palettized, x = var_1461_cast_fp16)[name = string("linear_20")]; tensor var_1529 = const()[name = string("op_1529"), val = tensor([1, 1, 256, 1])]; tensor var_1530 = reshape(shape = var_1529, x = var_1524)[name = string("op_1530")]; tensor var_1535 = const()[name = string("op_1535"), val = tensor([0, 1, 3, 2])]; tensor var_1544 = linear(bias = linear_2_bias_0, weight = layers_2_self_attn_v_proj_weight_palettized, x = var_1461_cast_fp16)[name = string("linear_21")]; tensor var_1549 = const()[name = string("op_1549"), val = tensor([1, 1, 256, 1])]; tensor var_1550 = reshape(shape = var_1549, x = var_1544)[name = string("op_1550")]; tensor var_1555 = const()[name = string("op_1555"), val = tensor([0, 1, 3, 2])]; tensor var_1565 = const()[name = string("op_1565"), val = tensor([1, 1, 256])]; tensor var_1536 = transpose(perm = var_1535, x = var_1530)[name = string("transpose_58")]; tensor x_43 = reshape(shape = var_1565, x = var_1536)[name = string("x_43")]; int32 var_1571 = const()[name = string("op_1571"), val = int32(-1)]; fp16 const_28_promoted = const()[name = string("const_28_promoted"), val = fp16(-0x1p+0)]; tensor var_1573 = mul(x = x_43, y = const_28_promoted)[name = string("op_1573")]; bool input_71_interleave_0 = const()[name = string("input_71_interleave_0"), val = bool(false)]; tensor input_71 = concat(axis = var_1571, interleave = input_71_interleave_0, values = (x_43, var_1573))[name = string("input_71")]; tensor normed_69_axes_0 = const()[name = string("normed_69_axes_0"), val = tensor([-1])]; fp16 var_1568_to_fp16 = const()[name = string("op_1568_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_69_cast_fp16 = layer_norm(axes = normed_69_axes_0, epsilon = var_1568_to_fp16, x = input_71)[name = string("normed_69_cast_fp16")]; tensor var_1578_split_sizes_0 = const()[name = string("op_1578_split_sizes_0"), val = tensor([256, 256])]; int32 var_1578_axis_0 = const()[name = string("op_1578_axis_0"), val = int32(-1)]; tensor var_1578_0, tensor var_1578_1 = split(axis = var_1578_axis_0, split_sizes = var_1578_split_sizes_0, x = normed_69_cast_fp16)[name = string("op_1578")]; tensor var_1580 = mul(x = var_1578_0, y = layers_2_self_attn_k_norm_weight)[name = string("op_1580")]; tensor var_1585 = const()[name = string("op_1585"), val = tensor([1, 1, 1, 256])]; tensor q_21 = reshape(shape = var_1585, x = var_1580)[name = string("q_21")]; fp16 var_1587_promoted = const()[name = string("op_1587_promoted"), val = fp16(0x1p+1)]; tensor var_1556 = transpose(perm = var_1555, x = var_1550)[name = string("transpose_57")]; tensor var_1588 = pow(x = var_1556, y = var_1587_promoted)[name = string("op_1588")]; tensor var_1593_axes_0 = const()[name = string("op_1593_axes_0"), val = tensor([-1])]; bool var_1593_keep_dims_0 = const()[name = string("op_1593_keep_dims_0"), val = bool(true)]; tensor var_1593 = reduce_mean(axes = var_1593_axes_0, keep_dims = var_1593_keep_dims_0, x = var_1588)[name = string("op_1593")]; fp16 var_1595_to_fp16 = const()[name = string("op_1595_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_5_cast_fp16 = add(x = var_1593, y = var_1595_to_fp16)[name = string("mean_sq_5_cast_fp16")]; fp32 var_1597_epsilon_0 = const()[name = string("op_1597_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_1597_cast_fp16 = rsqrt(epsilon = var_1597_epsilon_0, x = mean_sq_5_cast_fp16)[name = string("op_1597_cast_fp16")]; tensor input_75_cast_fp16 = mul(x = var_1556, y = var_1597_cast_fp16)[name = string("input_75_cast_fp16")]; tensor var_1599_cast_fp16 = mul(x = q_21, y = cos_s)[name = string("op_1599_cast_fp16")]; tensor var_1600_split_sizes_0 = const()[name = string("op_1600_split_sizes_0"), val = tensor([128, 128])]; int32 var_1600_axis_0 = const()[name = string("op_1600_axis_0"), val = int32(-1)]; tensor var_1600_0, tensor var_1600_1 = split(axis = var_1600_axis_0, split_sizes = var_1600_split_sizes_0, x = q_21)[name = string("op_1600")]; fp16 const_29_promoted = const()[name = string("const_29_promoted"), val = fp16(-0x1p+0)]; tensor var_1602 = mul(x = var_1600_1, y = const_29_promoted)[name = string("op_1602")]; int32 var_1604 = const()[name = string("op_1604"), val = int32(-1)]; bool var_1605_interleave_0 = const()[name = string("op_1605_interleave_0"), val = bool(false)]; tensor var_1605 = concat(axis = var_1604, interleave = var_1605_interleave_0, values = (var_1602, var_1600_0))[name = string("op_1605")]; tensor var_1606_cast_fp16 = mul(x = var_1605, y = sin_s)[name = string("op_1606_cast_fp16")]; tensor input_73_cast_fp16 = add(x = var_1599_cast_fp16, y = var_1606_cast_fp16)[name = string("input_73_cast_fp16")]; tensor k_padded_5_pad_0 = const()[name = string("k_padded_5_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_5_mode_0 = const()[name = string("k_padded_5_mode_0"), val = string("constant")]; fp16 const_30_to_fp16 = const()[name = string("const_30_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_5_cast_fp16 = pad(constant_val = const_30_to_fp16, mode = k_padded_5_mode_0, pad = k_padded_5_pad_0, x = input_73_cast_fp16)[name = string("k_padded_5_cast_fp16")]; tensor v_padded_5_pad_0 = const()[name = string("v_padded_5_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_5_mode_0 = const()[name = string("v_padded_5_mode_0"), val = string("constant")]; fp16 const_31_to_fp16 = const()[name = string("const_31_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_5_cast_fp16 = pad(constant_val = const_31_to_fp16, mode = v_padded_5_mode_0, pad = v_padded_5_pad_0, x = input_75_cast_fp16)[name = string("v_padded_5_cast_fp16")]; tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([4])]; tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([5])]; int32 concat_26_axis_0 = const()[name = string("concat_26_axis_0"), val = int32(0)]; bool concat_26_interleave_0 = const()[name = string("concat_26_interleave_0"), val = bool(false)]; tensor concat_26 = concat(axis = concat_26_axis_0, interleave = concat_26_interleave_0, values = (expand_dims_24, expand_dims_25, ring_pos, expand_dims_27))[name = string("concat_26")]; tensor concat_27_values1_0 = const()[name = string("concat_27_values1_0"), val = tensor([0])]; tensor concat_27_values3_0 = const()[name = string("concat_27_values3_0"), val = tensor([0])]; int32 concat_27_axis_0 = const()[name = string("concat_27_axis_0"), val = int32(0)]; bool concat_27_interleave_0 = const()[name = string("concat_27_interleave_0"), val = bool(false)]; tensor concat_27 = concat(axis = concat_27_axis_0, interleave = concat_27_interleave_0, values = (expand_dims_28, concat_27_values1_0, var_725, concat_27_values3_0))[name = string("concat_27")]; tensor kv_cache_sliding_internal_tensor_assign_5_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_5_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_5_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_26, begin_mask = kv_cache_sliding_internal_tensor_assign_5_begin_mask_0, end = concat_27, end_mask = kv_cache_sliding_internal_tensor_assign_5_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_5_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_5_stride_0, update = k_padded_5_cast_fp16, x = coreml_update_state_19)[name = string("kv_cache_sliding_internal_tensor_assign_5_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_5_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_4_write_state")]; tensor coreml_update_state_20 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_4")]; tensor expand_dims_30 = const()[name = string("expand_dims_30"), val = tensor([5])]; tensor expand_dims_31 = const()[name = string("expand_dims_31"), val = tensor([0])]; tensor expand_dims_33 = const()[name = string("expand_dims_33"), val = tensor([0])]; tensor expand_dims_34 = const()[name = string("expand_dims_34"), val = tensor([6])]; int32 concat_30_axis_0 = const()[name = string("concat_30_axis_0"), val = int32(0)]; bool concat_30_interleave_0 = const()[name = string("concat_30_interleave_0"), val = bool(false)]; tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (expand_dims_30, expand_dims_31, ring_pos, expand_dims_33))[name = string("concat_30")]; tensor concat_31_values1_0 = const()[name = string("concat_31_values1_0"), val = tensor([0])]; tensor concat_31_values3_0 = const()[name = string("concat_31_values3_0"), val = tensor([0])]; int32 concat_31_axis_0 = const()[name = string("concat_31_axis_0"), val = int32(0)]; bool concat_31_interleave_0 = const()[name = string("concat_31_interleave_0"), val = bool(false)]; tensor concat_31 = concat(axis = concat_31_axis_0, interleave = concat_31_interleave_0, values = (expand_dims_34, concat_31_values1_0, var_725, concat_31_values3_0))[name = string("concat_31")]; tensor kv_cache_sliding_internal_tensor_assign_6_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_6_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_6_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_30, begin_mask = kv_cache_sliding_internal_tensor_assign_6_begin_mask_0, end = concat_31, end_mask = kv_cache_sliding_internal_tensor_assign_6_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_6_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_6_stride_0, update = v_padded_5_cast_fp16, x = coreml_update_state_20)[name = string("kv_cache_sliding_internal_tensor_assign_6_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_6_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_5_write_state")]; tensor coreml_update_state_21 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_5")]; tensor var_1673_begin_0 = const()[name = string("op_1673_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_1673_end_0 = const()[name = string("op_1673_end_0"), val = tensor([5, 1, 512, 512])]; tensor var_1673_end_mask_0 = const()[name = string("op_1673_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1673_cast_fp16 = slice_by_index(begin = var_1673_begin_0, end = var_1673_end_0, end_mask = var_1673_end_mask_0, x = coreml_update_state_21)[name = string("op_1673_cast_fp16")]; tensor K_sliding_slice_5_begin_0 = const()[name = string("K_sliding_slice_5_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_sliding_slice_5_end_0 = const()[name = string("K_sliding_slice_5_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_sliding_slice_5_end_mask_0 = const()[name = string("K_sliding_slice_5_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_sliding_slice_5_cast_fp16 = slice_by_index(begin = K_sliding_slice_5_begin_0, end = K_sliding_slice_5_end_0, end_mask = K_sliding_slice_5_end_mask_0, x = var_1673_cast_fp16)[name = string("K_sliding_slice_5_cast_fp16")]; tensor var_1693_begin_0 = const()[name = string("op_1693_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_1693_end_0 = const()[name = string("op_1693_end_0"), val = tensor([6, 1, 512, 512])]; tensor var_1693_end_mask_0 = const()[name = string("op_1693_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1693_cast_fp16 = slice_by_index(begin = var_1693_begin_0, end = var_1693_end_0, end_mask = var_1693_end_mask_0, x = coreml_update_state_21)[name = string("op_1693_cast_fp16")]; tensor V_for_attn_5_begin_0 = const()[name = string("V_for_attn_5_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_5_end_0 = const()[name = string("V_for_attn_5_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_5_end_mask_0 = const()[name = string("V_for_attn_5_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_5_cast_fp16 = slice_by_index(begin = V_for_attn_5_begin_0, end = V_for_attn_5_end_0, end_mask = V_for_attn_5_end_mask_0, x = var_1693_cast_fp16)[name = string("V_for_attn_5_cast_fp16")]; tensor transpose_8_perm_0 = const()[name = string("transpose_8_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_4_reps_0 = const()[name = string("tile_4_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_8_cast_fp16 = transpose(perm = transpose_8_perm_0, x = K_sliding_slice_5_cast_fp16)[name = string("transpose_56")]; tensor tile_4_cast_fp16 = tile(reps = tile_4_reps_0, x = transpose_8_cast_fp16)[name = string("tile_4_cast_fp16")]; tensor concat_32 = const()[name = string("concat_32"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_8_cast_fp16 = reshape(shape = concat_32, x = tile_4_cast_fp16)[name = string("reshape_8_cast_fp16")]; tensor transpose_9_perm_0 = const()[name = string("transpose_9_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_33 = const()[name = string("concat_33"), val = tensor([-1, 1, 512, 256])]; tensor transpose_9_cast_fp16 = transpose(perm = transpose_9_perm_0, x = reshape_8_cast_fp16)[name = string("transpose_55")]; tensor reshape_9_cast_fp16 = reshape(shape = concat_33, x = transpose_9_cast_fp16)[name = string("reshape_9_cast_fp16")]; tensor transpose_34_perm_0 = const()[name = string("transpose_34_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_10_perm_0 = const()[name = string("transpose_10_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_5_reps_0 = const()[name = string("tile_5_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_10_cast_fp16 = transpose(perm = transpose_10_perm_0, x = V_for_attn_5_cast_fp16)[name = string("transpose_54")]; tensor tile_5_cast_fp16 = tile(reps = tile_5_reps_0, x = transpose_10_cast_fp16)[name = string("tile_5_cast_fp16")]; tensor concat_34 = const()[name = string("concat_34"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_10_cast_fp16 = reshape(shape = concat_34, x = tile_5_cast_fp16)[name = string("reshape_10_cast_fp16")]; tensor transpose_11_perm_0 = const()[name = string("transpose_11_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_35 = const()[name = string("concat_35"), val = tensor([-1, 1, 512, 256])]; tensor transpose_11_cast_fp16 = transpose(perm = transpose_11_perm_0, x = reshape_10_cast_fp16)[name = string("transpose_53")]; tensor reshape_11_cast_fp16 = reshape(shape = concat_35, x = transpose_11_cast_fp16)[name = string("reshape_11_cast_fp16")]; tensor V_expanded_5_perm_0 = const()[name = string("V_expanded_5_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_9_transpose_x_0 = const()[name = string("attn_weights_9_transpose_x_0"), val = bool(false)]; bool attn_weights_9_transpose_y_0 = const()[name = string("attn_weights_9_transpose_y_0"), val = bool(false)]; tensor transpose_34_cast_fp16 = transpose(perm = transpose_34_perm_0, x = reshape_9_cast_fp16)[name = string("transpose_52")]; tensor attn_weights_9_cast_fp16 = matmul(transpose_x = attn_weights_9_transpose_x_0, transpose_y = attn_weights_9_transpose_y_0, x = q_23_cast_fp16, y = transpose_34_cast_fp16)[name = string("attn_weights_9_cast_fp16")]; tensor x_47_cast_fp16 = add(x = attn_weights_9_cast_fp16, y = causal_mask_sliding)[name = string("x_47_cast_fp16")]; tensor reduce_max_2_axes_0 = const()[name = string("reduce_max_2_axes_0"), val = tensor([-1])]; bool reduce_max_2_keep_dims_0 = const()[name = string("reduce_max_2_keep_dims_0"), val = bool(true)]; tensor reduce_max_2 = reduce_max(axes = reduce_max_2_axes_0, keep_dims = reduce_max_2_keep_dims_0, x = x_47_cast_fp16)[name = string("reduce_max_2")]; tensor var_1738 = sub(x = x_47_cast_fp16, y = reduce_max_2)[name = string("op_1738")]; tensor var_1744 = exp(x = var_1738)[name = string("op_1744")]; tensor var_1754_axes_0 = const()[name = string("op_1754_axes_0"), val = tensor([-1])]; bool var_1754_keep_dims_0 = const()[name = string("op_1754_keep_dims_0"), val = bool(true)]; tensor var_1754 = reduce_sum(axes = var_1754_axes_0, keep_dims = var_1754_keep_dims_0, x = var_1744)[name = string("op_1754")]; tensor var_1760_cast_fp16 = real_div(x = var_1744, y = var_1754)[name = string("op_1760_cast_fp16")]; bool attn_output_9_transpose_x_0 = const()[name = string("attn_output_9_transpose_x_0"), val = bool(false)]; bool attn_output_9_transpose_y_0 = const()[name = string("attn_output_9_transpose_y_0"), val = bool(false)]; tensor V_expanded_5_cast_fp16 = transpose(perm = V_expanded_5_perm_0, x = reshape_11_cast_fp16)[name = string("transpose_51")]; tensor attn_output_9_cast_fp16 = matmul(transpose_x = attn_output_9_transpose_x_0, transpose_y = attn_output_9_transpose_y_0, x = var_1760_cast_fp16, y = V_expanded_5_cast_fp16)[name = string("attn_output_9_cast_fp16")]; tensor var_1771 = const()[name = string("op_1771"), val = tensor([0, 2, 1, 3])]; tensor var_1778 = const()[name = string("op_1778"), val = tensor([1, 1, -1])]; tensor var_1772_cast_fp16 = transpose(perm = var_1771, x = attn_output_9_cast_fp16)[name = string("transpose_50")]; tensor input_77_cast_fp16 = reshape(shape = var_1778, x = var_1772_cast_fp16)[name = string("input_77_cast_fp16")]; tensor layers_2_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(143193280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144766208))))[name = string("layers_2_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_22_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_2_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_77_cast_fp16)[name = string("linear_22_cast_fp16")]; int32 var_1787 = const()[name = string("op_1787"), val = int32(-1)]; fp16 const_32_promoted_to_fp16 = const()[name = string("const_32_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1789_cast_fp16 = mul(x = linear_22_cast_fp16, y = const_32_promoted_to_fp16)[name = string("op_1789_cast_fp16")]; bool input_79_interleave_0 = const()[name = string("input_79_interleave_0"), val = bool(false)]; tensor input_79_cast_fp16 = concat(axis = var_1787, interleave = input_79_interleave_0, values = (linear_22_cast_fp16, var_1789_cast_fp16))[name = string("input_79_cast_fp16")]; tensor normed_73_axes_0 = const()[name = string("normed_73_axes_0"), val = tensor([-1])]; fp16 var_1784_to_fp16 = const()[name = string("op_1784_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_73_cast_fp16 = layer_norm(axes = normed_73_axes_0, epsilon = var_1784_to_fp16, x = input_79_cast_fp16)[name = string("normed_73_cast_fp16")]; tensor var_1794_split_sizes_0 = const()[name = string("op_1794_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1794_axis_0 = const()[name = string("op_1794_axis_0"), val = int32(-1)]; tensor var_1794_cast_fp16_0, tensor var_1794_cast_fp16_1 = split(axis = var_1794_axis_0, split_sizes = var_1794_split_sizes_0, x = normed_73_cast_fp16)[name = string("op_1794_cast_fp16")]; tensor layers_2_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_2_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144767808)))]; tensor attn_output_11_cast_fp16 = mul(x = var_1794_cast_fp16_0, y = layers_2_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_11_cast_fp16")]; tensor x_53_cast_fp16 = add(x = x_39_cast_fp16, y = attn_output_11_cast_fp16)[name = string("x_53_cast_fp16")]; int32 var_1803 = const()[name = string("op_1803"), val = int32(-1)]; fp16 const_33_promoted_to_fp16 = const()[name = string("const_33_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1805_cast_fp16 = mul(x = x_53_cast_fp16, y = const_33_promoted_to_fp16)[name = string("op_1805_cast_fp16")]; bool input_81_interleave_0 = const()[name = string("input_81_interleave_0"), val = bool(false)]; tensor input_81_cast_fp16 = concat(axis = var_1803, interleave = input_81_interleave_0, values = (x_53_cast_fp16, var_1805_cast_fp16))[name = string("input_81_cast_fp16")]; tensor normed_77_axes_0 = const()[name = string("normed_77_axes_0"), val = tensor([-1])]; fp16 var_1800_to_fp16 = const()[name = string("op_1800_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_77_cast_fp16 = layer_norm(axes = normed_77_axes_0, epsilon = var_1800_to_fp16, x = input_81_cast_fp16)[name = string("normed_77_cast_fp16")]; tensor var_1810_split_sizes_0 = const()[name = string("op_1810_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1810_axis_0 = const()[name = string("op_1810_axis_0"), val = int32(-1)]; tensor var_1810_cast_fp16_0, tensor var_1810_cast_fp16_1 = split(axis = var_1810_axis_0, split_sizes = var_1810_split_sizes_0, x = normed_77_cast_fp16)[name = string("op_1810_cast_fp16")]; tensor layers_2_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_2_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144770944)))]; tensor var_1812_cast_fp16 = mul(x = var_1810_cast_fp16_0, y = layers_2_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("op_1812_cast_fp16")]; tensor gate_9 = linear(bias = linear_5_bias_0, weight = layers_2_mlp_gate_proj_weight_palettized, x = var_1812_cast_fp16)[name = string("linear_23")]; tensor up_5 = linear(bias = linear_5_bias_0, weight = layers_2_mlp_up_proj_weight_palettized, x = var_1812_cast_fp16)[name = string("linear_24")]; string gate_11_mode_0 = const()[name = string("gate_11_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_11 = gelu(mode = gate_11_mode_0, x = gate_9)[name = string("gate_11")]; tensor input_85 = mul(x = gate_11, y = up_5)[name = string("input_85")]; tensor x_55 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_2_mlp_down_proj_weight_palettized, x = input_85)[name = string("linear_25")]; int32 var_1834 = const()[name = string("op_1834"), val = int32(-1)]; fp16 const_34_promoted = const()[name = string("const_34_promoted"), val = fp16(-0x1p+0)]; tensor var_1836 = mul(x = x_55, y = const_34_promoted)[name = string("op_1836")]; bool input_87_interleave_0 = const()[name = string("input_87_interleave_0"), val = bool(false)]; tensor input_87 = concat(axis = var_1834, interleave = input_87_interleave_0, values = (x_55, var_1836))[name = string("input_87")]; tensor normed_81_axes_0 = const()[name = string("normed_81_axes_0"), val = tensor([-1])]; fp16 var_1831_to_fp16 = const()[name = string("op_1831_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_81_cast_fp16 = layer_norm(axes = normed_81_axes_0, epsilon = var_1831_to_fp16, x = input_87)[name = string("normed_81_cast_fp16")]; tensor var_1841_split_sizes_0 = const()[name = string("op_1841_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1841_axis_0 = const()[name = string("op_1841_axis_0"), val = int32(-1)]; tensor var_1841_0, tensor var_1841_1 = split(axis = var_1841_axis_0, split_sizes = var_1841_split_sizes_0, x = normed_81_cast_fp16)[name = string("op_1841")]; tensor hidden_states_19 = mul(x = var_1841_0, y = layers_2_post_feedforward_layernorm_weight)[name = string("hidden_states_19")]; tensor hidden_states_21_cast_fp16 = add(x = x_53_cast_fp16, y = hidden_states_19)[name = string("hidden_states_21_cast_fp16")]; tensor per_layer_slice_5_begin_0 = const()[name = string("per_layer_slice_5_begin_0"), val = tensor([0, 0, 512])]; tensor per_layer_slice_5_end_0 = const()[name = string("per_layer_slice_5_end_0"), val = tensor([1, 1, 768])]; tensor per_layer_slice_5_end_mask_0 = const()[name = string("per_layer_slice_5_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_5_cast_fp16 = slice_by_index(begin = per_layer_slice_5_begin_0, end = per_layer_slice_5_end_0, end_mask = per_layer_slice_5_end_mask_0, x = per_layer_combined_out)[name = string("per_layer_slice_5_cast_fp16")]; tensor gated_9 = linear(bias = linear_2_bias_0, weight = layers_2_per_layer_input_gate_weight_palettized, x = hidden_states_21_cast_fp16)[name = string("linear_26")]; string gated_11_mode_0 = const()[name = string("gated_11_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_11 = gelu(mode = gated_11_mode_0, x = gated_9)[name = string("gated_11")]; tensor input_91_cast_fp16 = mul(x = gated_11, y = per_layer_slice_5_cast_fp16)[name = string("input_91_cast_fp16")]; tensor layers_2_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144774080))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144970752))))[name = string("layers_2_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_27_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_2_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_91_cast_fp16)[name = string("linear_27_cast_fp16")]; int32 var_1879 = const()[name = string("op_1879"), val = int32(-1)]; fp16 const_35_promoted_to_fp16 = const()[name = string("const_35_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1881_cast_fp16 = mul(x = linear_27_cast_fp16, y = const_35_promoted_to_fp16)[name = string("op_1881_cast_fp16")]; bool input_93_interleave_0 = const()[name = string("input_93_interleave_0"), val = bool(false)]; tensor input_93_cast_fp16 = concat(axis = var_1879, interleave = input_93_interleave_0, values = (linear_27_cast_fp16, var_1881_cast_fp16))[name = string("input_93_cast_fp16")]; tensor normed_85_axes_0 = const()[name = string("normed_85_axes_0"), val = tensor([-1])]; fp16 var_1876_to_fp16 = const()[name = string("op_1876_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_85_cast_fp16 = layer_norm(axes = normed_85_axes_0, epsilon = var_1876_to_fp16, x = input_93_cast_fp16)[name = string("normed_85_cast_fp16")]; tensor var_1886_split_sizes_0 = const()[name = string("op_1886_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1886_axis_0 = const()[name = string("op_1886_axis_0"), val = int32(-1)]; tensor var_1886_cast_fp16_0, tensor var_1886_cast_fp16_1 = split(axis = var_1886_axis_0, split_sizes = var_1886_split_sizes_0, x = normed_85_cast_fp16)[name = string("op_1886_cast_fp16")]; tensor layers_2_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_2_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144972352)))]; tensor hidden_states_23_cast_fp16 = mul(x = var_1886_cast_fp16_0, y = layers_2_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_23_cast_fp16")]; tensor hidden_states_25_cast_fp16 = add(x = hidden_states_21_cast_fp16, y = hidden_states_23_cast_fp16)[name = string("hidden_states_25_cast_fp16")]; tensor const_36_promoted_to_fp16 = const()[name = string("const_36_promoted_to_fp16"), val = tensor([0x1.96p-1])]; tensor x_59_cast_fp16 = mul(x = hidden_states_25_cast_fp16, y = const_36_promoted_to_fp16)[name = string("x_59_cast_fp16")]; int32 var_1901 = const()[name = string("op_1901"), val = int32(-1)]; fp16 const_37_promoted_to_fp16 = const()[name = string("const_37_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1903_cast_fp16 = mul(x = x_59_cast_fp16, y = const_37_promoted_to_fp16)[name = string("op_1903_cast_fp16")]; bool input_95_interleave_0 = const()[name = string("input_95_interleave_0"), val = bool(false)]; tensor input_95_cast_fp16 = concat(axis = var_1901, interleave = input_95_interleave_0, values = (x_59_cast_fp16, var_1903_cast_fp16))[name = string("input_95_cast_fp16")]; tensor normed_89_axes_0 = const()[name = string("normed_89_axes_0"), val = tensor([-1])]; fp16 var_1898_to_fp16 = const()[name = string("op_1898_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_89_cast_fp16 = layer_norm(axes = normed_89_axes_0, epsilon = var_1898_to_fp16, x = input_95_cast_fp16)[name = string("normed_89_cast_fp16")]; tensor var_1908_split_sizes_0 = const()[name = string("op_1908_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1908_axis_0 = const()[name = string("op_1908_axis_0"), val = int32(-1)]; tensor var_1908_cast_fp16_0, tensor var_1908_cast_fp16_1 = split(axis = var_1908_axis_0, split_sizes = var_1908_split_sizes_0, x = normed_89_cast_fp16)[name = string("op_1908_cast_fp16")]; tensor layers_3_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_3_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144975488)))]; tensor var_1910_cast_fp16 = mul(x = var_1908_cast_fp16_0, y = layers_3_input_layernorm_weight_promoted_to_fp16)[name = string("op_1910_cast_fp16")]; tensor var_1918 = linear(bias = linear_1_bias_0, weight = layers_3_self_attn_q_proj_weight_palettized, x = var_1910_cast_fp16)[name = string("linear_28")]; tensor var_1923 = const()[name = string("op_1923"), val = tensor([1, 8, 256, 1])]; tensor var_1924 = reshape(shape = var_1923, x = var_1918)[name = string("op_1924")]; tensor var_1929 = const()[name = string("op_1929"), val = tensor([0, 1, 3, 2])]; tensor var_1939 = const()[name = string("op_1939"), val = tensor([1, 8, 256])]; tensor var_1930 = transpose(perm = var_1929, x = var_1924)[name = string("transpose_49")]; tensor x_61 = reshape(shape = var_1939, x = var_1930)[name = string("x_61")]; int32 var_1945 = const()[name = string("op_1945"), val = int32(-1)]; fp16 const_38_promoted = const()[name = string("const_38_promoted"), val = fp16(-0x1p+0)]; tensor var_1947 = mul(x = x_61, y = const_38_promoted)[name = string("op_1947")]; bool input_99_interleave_0 = const()[name = string("input_99_interleave_0"), val = bool(false)]; tensor input_99 = concat(axis = var_1945, interleave = input_99_interleave_0, values = (x_61, var_1947))[name = string("input_99")]; tensor normed_93_axes_0 = const()[name = string("normed_93_axes_0"), val = tensor([-1])]; fp16 var_1942_to_fp16 = const()[name = string("op_1942_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_93_cast_fp16 = layer_norm(axes = normed_93_axes_0, epsilon = var_1942_to_fp16, x = input_99)[name = string("normed_93_cast_fp16")]; tensor var_1952_split_sizes_0 = const()[name = string("op_1952_split_sizes_0"), val = tensor([256, 256])]; int32 var_1952_axis_0 = const()[name = string("op_1952_axis_0"), val = int32(-1)]; tensor var_1952_0, tensor var_1952_1 = split(axis = var_1952_axis_0, split_sizes = var_1952_split_sizes_0, x = normed_93_cast_fp16)[name = string("op_1952")]; tensor var_1954 = mul(x = var_1952_0, y = layers_3_self_attn_q_norm_weight)[name = string("op_1954")]; tensor var_1959 = const()[name = string("op_1959"), val = tensor([1, 8, 1, 256])]; tensor q_27 = reshape(shape = var_1959, x = var_1954)[name = string("q_27")]; tensor var_1961_cast_fp16 = mul(x = q_27, y = cos_s)[name = string("op_1961_cast_fp16")]; tensor var_1962_split_sizes_0 = const()[name = string("op_1962_split_sizes_0"), val = tensor([128, 128])]; int32 var_1962_axis_0 = const()[name = string("op_1962_axis_0"), val = int32(-1)]; tensor var_1962_0, tensor var_1962_1 = split(axis = var_1962_axis_0, split_sizes = var_1962_split_sizes_0, x = q_27)[name = string("op_1962")]; fp16 const_39_promoted = const()[name = string("const_39_promoted"), val = fp16(-0x1p+0)]; tensor var_1964 = mul(x = var_1962_1, y = const_39_promoted)[name = string("op_1964")]; int32 var_1966 = const()[name = string("op_1966"), val = int32(-1)]; bool var_1967_interleave_0 = const()[name = string("op_1967_interleave_0"), val = bool(false)]; tensor var_1967 = concat(axis = var_1966, interleave = var_1967_interleave_0, values = (var_1964, var_1962_0))[name = string("op_1967")]; tensor var_1968_cast_fp16 = mul(x = var_1967, y = sin_s)[name = string("op_1968_cast_fp16")]; tensor q_31_cast_fp16 = add(x = var_1961_cast_fp16, y = var_1968_cast_fp16)[name = string("q_31_cast_fp16")]; tensor var_1973 = linear(bias = linear_2_bias_0, weight = layers_3_self_attn_k_proj_weight_palettized, x = var_1910_cast_fp16)[name = string("linear_29")]; tensor var_1978 = const()[name = string("op_1978"), val = tensor([1, 1, 256, 1])]; tensor var_1979 = reshape(shape = var_1978, x = var_1973)[name = string("op_1979")]; tensor var_1984 = const()[name = string("op_1984"), val = tensor([0, 1, 3, 2])]; tensor var_1993 = linear(bias = linear_2_bias_0, weight = layers_3_self_attn_v_proj_weight_palettized, x = var_1910_cast_fp16)[name = string("linear_30")]; tensor var_1998 = const()[name = string("op_1998"), val = tensor([1, 1, 256, 1])]; tensor var_1999 = reshape(shape = var_1998, x = var_1993)[name = string("op_1999")]; tensor var_2004 = const()[name = string("op_2004"), val = tensor([0, 1, 3, 2])]; tensor var_2014 = const()[name = string("op_2014"), val = tensor([1, 1, 256])]; tensor var_1985 = transpose(perm = var_1984, x = var_1979)[name = string("transpose_48")]; tensor x_63 = reshape(shape = var_2014, x = var_1985)[name = string("x_63")]; int32 var_2020 = const()[name = string("op_2020"), val = int32(-1)]; fp16 const_40_promoted = const()[name = string("const_40_promoted"), val = fp16(-0x1p+0)]; tensor var_2022 = mul(x = x_63, y = const_40_promoted)[name = string("op_2022")]; bool input_101_interleave_0 = const()[name = string("input_101_interleave_0"), val = bool(false)]; tensor input_101 = concat(axis = var_2020, interleave = input_101_interleave_0, values = (x_63, var_2022))[name = string("input_101")]; tensor normed_97_axes_0 = const()[name = string("normed_97_axes_0"), val = tensor([-1])]; fp16 var_2017_to_fp16 = const()[name = string("op_2017_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_97_cast_fp16 = layer_norm(axes = normed_97_axes_0, epsilon = var_2017_to_fp16, x = input_101)[name = string("normed_97_cast_fp16")]; tensor var_2027_split_sizes_0 = const()[name = string("op_2027_split_sizes_0"), val = tensor([256, 256])]; int32 var_2027_axis_0 = const()[name = string("op_2027_axis_0"), val = int32(-1)]; tensor var_2027_0, tensor var_2027_1 = split(axis = var_2027_axis_0, split_sizes = var_2027_split_sizes_0, x = normed_97_cast_fp16)[name = string("op_2027")]; tensor var_2029 = mul(x = var_2027_0, y = layers_3_self_attn_k_norm_weight)[name = string("op_2029")]; tensor var_2034 = const()[name = string("op_2034"), val = tensor([1, 1, 1, 256])]; tensor q_29 = reshape(shape = var_2034, x = var_2029)[name = string("q_29")]; fp16 var_2036_promoted = const()[name = string("op_2036_promoted"), val = fp16(0x1p+1)]; tensor var_2005 = transpose(perm = var_2004, x = var_1999)[name = string("transpose_47")]; tensor var_2037 = pow(x = var_2005, y = var_2036_promoted)[name = string("op_2037")]; tensor var_2042_axes_0 = const()[name = string("op_2042_axes_0"), val = tensor([-1])]; bool var_2042_keep_dims_0 = const()[name = string("op_2042_keep_dims_0"), val = bool(true)]; tensor var_2042 = reduce_mean(axes = var_2042_axes_0, keep_dims = var_2042_keep_dims_0, x = var_2037)[name = string("op_2042")]; fp16 var_2044_to_fp16 = const()[name = string("op_2044_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_7_cast_fp16 = add(x = var_2042, y = var_2044_to_fp16)[name = string("mean_sq_7_cast_fp16")]; fp32 var_2046_epsilon_0 = const()[name = string("op_2046_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_2046_cast_fp16 = rsqrt(epsilon = var_2046_epsilon_0, x = mean_sq_7_cast_fp16)[name = string("op_2046_cast_fp16")]; tensor input_105_cast_fp16 = mul(x = var_2005, y = var_2046_cast_fp16)[name = string("input_105_cast_fp16")]; tensor var_2048_cast_fp16 = mul(x = q_29, y = cos_s)[name = string("op_2048_cast_fp16")]; tensor var_2049_split_sizes_0 = const()[name = string("op_2049_split_sizes_0"), val = tensor([128, 128])]; int32 var_2049_axis_0 = const()[name = string("op_2049_axis_0"), val = int32(-1)]; tensor var_2049_0, tensor var_2049_1 = split(axis = var_2049_axis_0, split_sizes = var_2049_split_sizes_0, x = q_29)[name = string("op_2049")]; fp16 const_41_promoted = const()[name = string("const_41_promoted"), val = fp16(-0x1p+0)]; tensor var_2051 = mul(x = var_2049_1, y = const_41_promoted)[name = string("op_2051")]; int32 var_2053 = const()[name = string("op_2053"), val = int32(-1)]; bool var_2054_interleave_0 = const()[name = string("op_2054_interleave_0"), val = bool(false)]; tensor var_2054 = concat(axis = var_2053, interleave = var_2054_interleave_0, values = (var_2051, var_2049_0))[name = string("op_2054")]; tensor var_2055_cast_fp16 = mul(x = var_2054, y = sin_s)[name = string("op_2055_cast_fp16")]; tensor input_103_cast_fp16 = add(x = var_2048_cast_fp16, y = var_2055_cast_fp16)[name = string("input_103_cast_fp16")]; tensor k_padded_7_pad_0 = const()[name = string("k_padded_7_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_7_mode_0 = const()[name = string("k_padded_7_mode_0"), val = string("constant")]; fp16 const_42_to_fp16 = const()[name = string("const_42_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_7_cast_fp16 = pad(constant_val = const_42_to_fp16, mode = k_padded_7_mode_0, pad = k_padded_7_pad_0, x = input_103_cast_fp16)[name = string("k_padded_7_cast_fp16")]; tensor v_padded_7_pad_0 = const()[name = string("v_padded_7_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_7_mode_0 = const()[name = string("v_padded_7_mode_0"), val = string("constant")]; fp16 const_43_to_fp16 = const()[name = string("const_43_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_7_cast_fp16 = pad(constant_val = const_43_to_fp16, mode = v_padded_7_mode_0, pad = v_padded_7_pad_0, x = input_105_cast_fp16)[name = string("v_padded_7_cast_fp16")]; tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([6])]; tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([7])]; int32 concat_38_axis_0 = const()[name = string("concat_38_axis_0"), val = int32(0)]; bool concat_38_interleave_0 = const()[name = string("concat_38_interleave_0"), val = bool(false)]; tensor concat_38 = concat(axis = concat_38_axis_0, interleave = concat_38_interleave_0, values = (expand_dims_36, expand_dims_37, ring_pos, expand_dims_39))[name = string("concat_38")]; tensor concat_39_values1_0 = const()[name = string("concat_39_values1_0"), val = tensor([0])]; tensor concat_39_values3_0 = const()[name = string("concat_39_values3_0"), val = tensor([0])]; int32 concat_39_axis_0 = const()[name = string("concat_39_axis_0"), val = int32(0)]; bool concat_39_interleave_0 = const()[name = string("concat_39_interleave_0"), val = bool(false)]; tensor concat_39 = concat(axis = concat_39_axis_0, interleave = concat_39_interleave_0, values = (expand_dims_40, concat_39_values1_0, var_725, concat_39_values3_0))[name = string("concat_39")]; tensor kv_cache_sliding_internal_tensor_assign_7_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_7_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_7_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_38, begin_mask = kv_cache_sliding_internal_tensor_assign_7_begin_mask_0, end = concat_39, end_mask = kv_cache_sliding_internal_tensor_assign_7_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_7_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_7_stride_0, update = k_padded_7_cast_fp16, x = coreml_update_state_21)[name = string("kv_cache_sliding_internal_tensor_assign_7_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_7_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_6_write_state")]; tensor coreml_update_state_22 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_6")]; tensor expand_dims_42 = const()[name = string("expand_dims_42"), val = tensor([7])]; tensor expand_dims_43 = const()[name = string("expand_dims_43"), val = tensor([0])]; tensor expand_dims_45 = const()[name = string("expand_dims_45"), val = tensor([0])]; tensor expand_dims_46 = const()[name = string("expand_dims_46"), val = tensor([8])]; int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_42, expand_dims_43, ring_pos, expand_dims_45))[name = string("concat_42")]; tensor concat_43_values1_0 = const()[name = string("concat_43_values1_0"), val = tensor([0])]; tensor concat_43_values3_0 = const()[name = string("concat_43_values3_0"), val = tensor([0])]; int32 concat_43_axis_0 = const()[name = string("concat_43_axis_0"), val = int32(0)]; bool concat_43_interleave_0 = const()[name = string("concat_43_interleave_0"), val = bool(false)]; tensor concat_43 = concat(axis = concat_43_axis_0, interleave = concat_43_interleave_0, values = (expand_dims_46, concat_43_values1_0, var_725, concat_43_values3_0))[name = string("concat_43")]; tensor kv_cache_sliding_internal_tensor_assign_8_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_8_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_8_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_42, begin_mask = kv_cache_sliding_internal_tensor_assign_8_begin_mask_0, end = concat_43, end_mask = kv_cache_sliding_internal_tensor_assign_8_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_8_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_8_stride_0, update = v_padded_7_cast_fp16, x = coreml_update_state_22)[name = string("kv_cache_sliding_internal_tensor_assign_8_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_8_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_7_write_state")]; tensor coreml_update_state_23 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_7")]; tensor var_2122_begin_0 = const()[name = string("op_2122_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_2122_end_0 = const()[name = string("op_2122_end_0"), val = tensor([7, 1, 512, 512])]; tensor var_2122_end_mask_0 = const()[name = string("op_2122_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2122_cast_fp16 = slice_by_index(begin = var_2122_begin_0, end = var_2122_end_0, end_mask = var_2122_end_mask_0, x = coreml_update_state_23)[name = string("op_2122_cast_fp16")]; tensor K_sliding_slice_7_begin_0 = const()[name = string("K_sliding_slice_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_sliding_slice_7_end_0 = const()[name = string("K_sliding_slice_7_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_sliding_slice_7_end_mask_0 = const()[name = string("K_sliding_slice_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_sliding_slice_7_cast_fp16 = slice_by_index(begin = K_sliding_slice_7_begin_0, end = K_sliding_slice_7_end_0, end_mask = K_sliding_slice_7_end_mask_0, x = var_2122_cast_fp16)[name = string("K_sliding_slice_7_cast_fp16")]; tensor var_2142_begin_0 = const()[name = string("op_2142_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_2142_end_0 = const()[name = string("op_2142_end_0"), val = tensor([8, 1, 512, 512])]; tensor var_2142_end_mask_0 = const()[name = string("op_2142_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2142_cast_fp16 = slice_by_index(begin = var_2142_begin_0, end = var_2142_end_0, end_mask = var_2142_end_mask_0, x = coreml_update_state_23)[name = string("op_2142_cast_fp16")]; tensor V_for_attn_7_begin_0 = const()[name = string("V_for_attn_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_7_end_0 = const()[name = string("V_for_attn_7_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_7_end_mask_0 = const()[name = string("V_for_attn_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_7_cast_fp16 = slice_by_index(begin = V_for_attn_7_begin_0, end = V_for_attn_7_end_0, end_mask = V_for_attn_7_end_mask_0, x = var_2142_cast_fp16)[name = string("V_for_attn_7_cast_fp16")]; tensor transpose_12_perm_0 = const()[name = string("transpose_12_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_6_reps_0 = const()[name = string("tile_6_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_12_cast_fp16 = transpose(perm = transpose_12_perm_0, x = K_sliding_slice_7_cast_fp16)[name = string("transpose_46")]; tensor tile_6_cast_fp16 = tile(reps = tile_6_reps_0, x = transpose_12_cast_fp16)[name = string("tile_6_cast_fp16")]; tensor concat_44 = const()[name = string("concat_44"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_12_cast_fp16 = reshape(shape = concat_44, x = tile_6_cast_fp16)[name = string("reshape_12_cast_fp16")]; tensor transpose_13_perm_0 = const()[name = string("transpose_13_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_45 = const()[name = string("concat_45"), val = tensor([-1, 1, 512, 256])]; tensor transpose_13_cast_fp16 = transpose(perm = transpose_13_perm_0, x = reshape_12_cast_fp16)[name = string("transpose_45")]; tensor reshape_13_cast_fp16 = reshape(shape = concat_45, x = transpose_13_cast_fp16)[name = string("reshape_13_cast_fp16")]; tensor transpose_35_perm_0 = const()[name = string("transpose_35_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_14_perm_0 = const()[name = string("transpose_14_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_7_reps_0 = const()[name = string("tile_7_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_14_cast_fp16 = transpose(perm = transpose_14_perm_0, x = V_for_attn_7_cast_fp16)[name = string("transpose_44")]; tensor tile_7_cast_fp16 = tile(reps = tile_7_reps_0, x = transpose_14_cast_fp16)[name = string("tile_7_cast_fp16")]; tensor concat_46 = const()[name = string("concat_46"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_14_cast_fp16 = reshape(shape = concat_46, x = tile_7_cast_fp16)[name = string("reshape_14_cast_fp16")]; tensor transpose_15_perm_0 = const()[name = string("transpose_15_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_47 = const()[name = string("concat_47"), val = tensor([-1, 1, 512, 256])]; tensor transpose_15_cast_fp16 = transpose(perm = transpose_15_perm_0, x = reshape_14_cast_fp16)[name = string("transpose_43")]; tensor reshape_15_cast_fp16 = reshape(shape = concat_47, x = transpose_15_cast_fp16)[name = string("reshape_15_cast_fp16")]; tensor V_expanded_7_perm_0 = const()[name = string("V_expanded_7_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_13_transpose_x_0 = const()[name = string("attn_weights_13_transpose_x_0"), val = bool(false)]; bool attn_weights_13_transpose_y_0 = const()[name = string("attn_weights_13_transpose_y_0"), val = bool(false)]; tensor transpose_35_cast_fp16 = transpose(perm = transpose_35_perm_0, x = reshape_13_cast_fp16)[name = string("transpose_42")]; tensor attn_weights_13_cast_fp16 = matmul(transpose_x = attn_weights_13_transpose_x_0, transpose_y = attn_weights_13_transpose_y_0, x = q_31_cast_fp16, y = transpose_35_cast_fp16)[name = string("attn_weights_13_cast_fp16")]; tensor x_67_cast_fp16 = add(x = attn_weights_13_cast_fp16, y = causal_mask_sliding)[name = string("x_67_cast_fp16")]; tensor reduce_max_3_axes_0 = const()[name = string("reduce_max_3_axes_0"), val = tensor([-1])]; bool reduce_max_3_keep_dims_0 = const()[name = string("reduce_max_3_keep_dims_0"), val = bool(true)]; tensor reduce_max_3 = reduce_max(axes = reduce_max_3_axes_0, keep_dims = reduce_max_3_keep_dims_0, x = x_67_cast_fp16)[name = string("reduce_max_3")]; tensor var_2187 = sub(x = x_67_cast_fp16, y = reduce_max_3)[name = string("op_2187")]; tensor var_2193 = exp(x = var_2187)[name = string("op_2193")]; tensor var_2203_axes_0 = const()[name = string("op_2203_axes_0"), val = tensor([-1])]; bool var_2203_keep_dims_0 = const()[name = string("op_2203_keep_dims_0"), val = bool(true)]; tensor var_2203 = reduce_sum(axes = var_2203_axes_0, keep_dims = var_2203_keep_dims_0, x = var_2193)[name = string("op_2203")]; tensor var_2209_cast_fp16 = real_div(x = var_2193, y = var_2203)[name = string("op_2209_cast_fp16")]; bool attn_output_13_transpose_x_0 = const()[name = string("attn_output_13_transpose_x_0"), val = bool(false)]; bool attn_output_13_transpose_y_0 = const()[name = string("attn_output_13_transpose_y_0"), val = bool(false)]; tensor V_expanded_7_cast_fp16 = transpose(perm = V_expanded_7_perm_0, x = reshape_15_cast_fp16)[name = string("transpose_41")]; tensor attn_output_13_cast_fp16 = matmul(transpose_x = attn_output_13_transpose_x_0, transpose_y = attn_output_13_transpose_y_0, x = var_2209_cast_fp16, y = V_expanded_7_cast_fp16)[name = string("attn_output_13_cast_fp16")]; tensor var_2220 = const()[name = string("op_2220"), val = tensor([0, 2, 1, 3])]; tensor var_2227 = const()[name = string("op_2227"), val = tensor([1, 1, -1])]; tensor var_2221_cast_fp16 = transpose(perm = var_2220, x = attn_output_13_cast_fp16)[name = string("transpose_40")]; tensor input_107_cast_fp16 = reshape(shape = var_2227, x = var_2221_cast_fp16)[name = string("input_107_cast_fp16")]; tensor layers_3_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144978624))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146551552))))[name = string("layers_3_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_31_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_3_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_107_cast_fp16)[name = string("linear_31_cast_fp16")]; int32 var_2236 = const()[name = string("op_2236"), val = int32(-1)]; fp16 const_44_promoted_to_fp16 = const()[name = string("const_44_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2238_cast_fp16 = mul(x = linear_31_cast_fp16, y = const_44_promoted_to_fp16)[name = string("op_2238_cast_fp16")]; bool input_109_interleave_0 = const()[name = string("input_109_interleave_0"), val = bool(false)]; tensor input_109_cast_fp16 = concat(axis = var_2236, interleave = input_109_interleave_0, values = (linear_31_cast_fp16, var_2238_cast_fp16))[name = string("input_109_cast_fp16")]; tensor normed_101_axes_0 = const()[name = string("normed_101_axes_0"), val = tensor([-1])]; fp16 var_2233_to_fp16 = const()[name = string("op_2233_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_101_cast_fp16 = layer_norm(axes = normed_101_axes_0, epsilon = var_2233_to_fp16, x = input_109_cast_fp16)[name = string("normed_101_cast_fp16")]; tensor var_2243_split_sizes_0 = const()[name = string("op_2243_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2243_axis_0 = const()[name = string("op_2243_axis_0"), val = int32(-1)]; tensor var_2243_cast_fp16_0, tensor var_2243_cast_fp16_1 = split(axis = var_2243_axis_0, split_sizes = var_2243_split_sizes_0, x = normed_101_cast_fp16)[name = string("op_2243_cast_fp16")]; tensor layers_3_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_3_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146553152)))]; tensor attn_output_15_cast_fp16 = mul(x = var_2243_cast_fp16_0, y = layers_3_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_15_cast_fp16")]; tensor x_73_cast_fp16 = add(x = x_59_cast_fp16, y = attn_output_15_cast_fp16)[name = string("x_73_cast_fp16")]; int32 var_2252 = const()[name = string("op_2252"), val = int32(-1)]; fp16 const_45_promoted_to_fp16 = const()[name = string("const_45_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2254_cast_fp16 = mul(x = x_73_cast_fp16, y = const_45_promoted_to_fp16)[name = string("op_2254_cast_fp16")]; bool input_111_interleave_0 = const()[name = string("input_111_interleave_0"), val = bool(false)]; tensor input_111_cast_fp16 = concat(axis = var_2252, interleave = input_111_interleave_0, values = (x_73_cast_fp16, var_2254_cast_fp16))[name = string("input_111_cast_fp16")]; tensor normed_105_axes_0 = const()[name = string("normed_105_axes_0"), val = tensor([-1])]; fp16 var_2249_to_fp16 = const()[name = string("op_2249_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_105_cast_fp16 = layer_norm(axes = normed_105_axes_0, epsilon = var_2249_to_fp16, x = input_111_cast_fp16)[name = string("normed_105_cast_fp16")]; tensor var_2259_split_sizes_0 = const()[name = string("op_2259_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2259_axis_0 = const()[name = string("op_2259_axis_0"), val = int32(-1)]; tensor var_2259_cast_fp16_0, tensor var_2259_cast_fp16_1 = split(axis = var_2259_axis_0, split_sizes = var_2259_split_sizes_0, x = normed_105_cast_fp16)[name = string("op_2259_cast_fp16")]; tensor layers_3_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_3_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146556288)))]; tensor var_2261_cast_fp16 = mul(x = var_2259_cast_fp16_0, y = layers_3_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("op_2261_cast_fp16")]; tensor gate_13 = linear(bias = linear_5_bias_0, weight = layers_3_mlp_gate_proj_weight_palettized, x = var_2261_cast_fp16)[name = string("linear_32")]; tensor up_7 = linear(bias = linear_5_bias_0, weight = layers_3_mlp_up_proj_weight_palettized, x = var_2261_cast_fp16)[name = string("linear_33")]; string gate_15_mode_0 = const()[name = string("gate_15_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_15 = gelu(mode = gate_15_mode_0, x = gate_13)[name = string("gate_15")]; tensor input_115 = mul(x = gate_15, y = up_7)[name = string("input_115")]; tensor x_75 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_3_mlp_down_proj_weight_palettized, x = input_115)[name = string("linear_34")]; int32 var_2283 = const()[name = string("op_2283"), val = int32(-1)]; fp16 const_46_promoted = const()[name = string("const_46_promoted"), val = fp16(-0x1p+0)]; tensor var_2285 = mul(x = x_75, y = const_46_promoted)[name = string("op_2285")]; bool input_117_interleave_0 = const()[name = string("input_117_interleave_0"), val = bool(false)]; tensor input_117 = concat(axis = var_2283, interleave = input_117_interleave_0, values = (x_75, var_2285))[name = string("input_117")]; tensor normed_109_axes_0 = const()[name = string("normed_109_axes_0"), val = tensor([-1])]; fp16 var_2280_to_fp16 = const()[name = string("op_2280_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_109_cast_fp16 = layer_norm(axes = normed_109_axes_0, epsilon = var_2280_to_fp16, x = input_117)[name = string("normed_109_cast_fp16")]; tensor var_2290_split_sizes_0 = const()[name = string("op_2290_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2290_axis_0 = const()[name = string("op_2290_axis_0"), val = int32(-1)]; tensor var_2290_0, tensor var_2290_1 = split(axis = var_2290_axis_0, split_sizes = var_2290_split_sizes_0, x = normed_109_cast_fp16)[name = string("op_2290")]; tensor hidden_states_27 = mul(x = var_2290_0, y = layers_3_post_feedforward_layernorm_weight)[name = string("hidden_states_27")]; tensor hidden_states_29_cast_fp16 = add(x = x_73_cast_fp16, y = hidden_states_27)[name = string("hidden_states_29_cast_fp16")]; tensor per_layer_slice_7_begin_0 = const()[name = string("per_layer_slice_7_begin_0"), val = tensor([0, 0, 768])]; tensor per_layer_slice_7_end_0 = const()[name = string("per_layer_slice_7_end_0"), val = tensor([1, 1, 1024])]; tensor per_layer_slice_7_end_mask_0 = const()[name = string("per_layer_slice_7_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_7_cast_fp16 = slice_by_index(begin = per_layer_slice_7_begin_0, end = per_layer_slice_7_end_0, end_mask = per_layer_slice_7_end_mask_0, x = per_layer_combined_out)[name = string("per_layer_slice_7_cast_fp16")]; tensor gated_13 = linear(bias = linear_2_bias_0, weight = layers_3_per_layer_input_gate_weight_palettized, x = hidden_states_29_cast_fp16)[name = string("linear_35")]; string gated_15_mode_0 = const()[name = string("gated_15_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_15 = gelu(mode = gated_15_mode_0, x = gated_13)[name = string("gated_15")]; tensor input_121_cast_fp16 = mul(x = gated_15, y = per_layer_slice_7_cast_fp16)[name = string("input_121_cast_fp16")]; tensor layers_3_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146559424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146756096))))[name = string("layers_3_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_36_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_3_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_121_cast_fp16)[name = string("linear_36_cast_fp16")]; int32 var_2328 = const()[name = string("op_2328"), val = int32(-1)]; fp16 const_47_promoted_to_fp16 = const()[name = string("const_47_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2330_cast_fp16 = mul(x = linear_36_cast_fp16, y = const_47_promoted_to_fp16)[name = string("op_2330_cast_fp16")]; bool input_123_interleave_0 = const()[name = string("input_123_interleave_0"), val = bool(false)]; tensor input_123_cast_fp16 = concat(axis = var_2328, interleave = input_123_interleave_0, values = (linear_36_cast_fp16, var_2330_cast_fp16))[name = string("input_123_cast_fp16")]; tensor normed_113_axes_0 = const()[name = string("normed_113_axes_0"), val = tensor([-1])]; fp16 var_2325_to_fp16 = const()[name = string("op_2325_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_113_cast_fp16 = layer_norm(axes = normed_113_axes_0, epsilon = var_2325_to_fp16, x = input_123_cast_fp16)[name = string("normed_113_cast_fp16")]; tensor var_2335_split_sizes_0 = const()[name = string("op_2335_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2335_axis_0 = const()[name = string("op_2335_axis_0"), val = int32(-1)]; tensor var_2335_cast_fp16_0, tensor var_2335_cast_fp16_1 = split(axis = var_2335_axis_0, split_sizes = var_2335_split_sizes_0, x = normed_113_cast_fp16)[name = string("op_2335_cast_fp16")]; tensor layers_3_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_3_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146757696)))]; tensor hidden_states_31_cast_fp16 = mul(x = var_2335_cast_fp16_0, y = layers_3_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_31_cast_fp16")]; tensor hidden_states_33_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = hidden_states_31_cast_fp16)[name = string("hidden_states_33_cast_fp16")]; tensor const_48_promoted_to_fp16 = const()[name = string("const_48_promoted_to_fp16"), val = tensor([0x1.26p-2])]; tensor x_79_cast_fp16 = mul(x = hidden_states_33_cast_fp16, y = const_48_promoted_to_fp16)[name = string("x_79_cast_fp16")]; int32 var_2350 = const()[name = string("op_2350"), val = int32(-1)]; fp16 const_49_promoted_to_fp16 = const()[name = string("const_49_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2352_cast_fp16 = mul(x = x_79_cast_fp16, y = const_49_promoted_to_fp16)[name = string("op_2352_cast_fp16")]; bool input_125_interleave_0 = const()[name = string("input_125_interleave_0"), val = bool(false)]; tensor input_125_cast_fp16 = concat(axis = var_2350, interleave = input_125_interleave_0, values = (x_79_cast_fp16, var_2352_cast_fp16))[name = string("input_125_cast_fp16")]; tensor normed_117_axes_0 = const()[name = string("normed_117_axes_0"), val = tensor([-1])]; fp16 var_2347_to_fp16 = const()[name = string("op_2347_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_117_cast_fp16 = layer_norm(axes = normed_117_axes_0, epsilon = var_2347_to_fp16, x = input_125_cast_fp16)[name = string("normed_117_cast_fp16")]; tensor var_2357_split_sizes_0 = const()[name = string("op_2357_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2357_axis_0 = const()[name = string("op_2357_axis_0"), val = int32(-1)]; tensor var_2357_cast_fp16_0, tensor var_2357_cast_fp16_1 = split(axis = var_2357_axis_0, split_sizes = var_2357_split_sizes_0, x = normed_117_cast_fp16)[name = string("op_2357_cast_fp16")]; tensor layers_4_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_4_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146760832)))]; tensor var_2359_cast_fp16 = mul(x = var_2357_cast_fp16_0, y = layers_4_input_layernorm_weight_promoted_to_fp16)[name = string("op_2359_cast_fp16")]; tensor linear_37_bias_0 = const()[name = string("linear_37_bias_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146763968)))]; tensor var_2367 = linear(bias = linear_37_bias_0, weight = layers_4_self_attn_q_proj_weight_palettized, x = var_2359_cast_fp16)[name = string("linear_37")]; tensor var_2372 = const()[name = string("op_2372"), val = tensor([1, 8, 512, 1])]; tensor var_2373 = reshape(shape = var_2372, x = var_2367)[name = string("op_2373")]; tensor var_2378 = const()[name = string("op_2378"), val = tensor([0, 1, 3, 2])]; tensor var_2388 = const()[name = string("op_2388"), val = tensor([1, 8, 512])]; tensor var_2379 = transpose(perm = var_2378, x = var_2373)[name = string("transpose_39")]; tensor x_81 = reshape(shape = var_2388, x = var_2379)[name = string("x_81")]; int32 var_2394 = const()[name = string("op_2394"), val = int32(-1)]; fp16 const_50_promoted = const()[name = string("const_50_promoted"), val = fp16(-0x1p+0)]; tensor var_2396 = mul(x = x_81, y = const_50_promoted)[name = string("op_2396")]; bool input_129_interleave_0 = const()[name = string("input_129_interleave_0"), val = bool(false)]; tensor input_129 = concat(axis = var_2394, interleave = input_129_interleave_0, values = (x_81, var_2396))[name = string("input_129")]; tensor normed_121_axes_0 = const()[name = string("normed_121_axes_0"), val = tensor([-1])]; fp16 var_2391_to_fp16 = const()[name = string("op_2391_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_121_cast_fp16 = layer_norm(axes = normed_121_axes_0, epsilon = var_2391_to_fp16, x = input_129)[name = string("normed_121_cast_fp16")]; tensor var_2401_split_sizes_0 = const()[name = string("op_2401_split_sizes_0"), val = tensor([512, 512])]; int32 var_2401_axis_0 = const()[name = string("op_2401_axis_0"), val = int32(-1)]; tensor var_2401_0, tensor var_2401_1 = split(axis = var_2401_axis_0, split_sizes = var_2401_split_sizes_0, x = normed_121_cast_fp16)[name = string("op_2401")]; tensor var_2403 = mul(x = var_2401_0, y = layers_4_self_attn_q_norm_weight)[name = string("op_2403")]; tensor var_2408 = const()[name = string("op_2408"), val = tensor([1, 8, 1, 512])]; tensor q_35 = reshape(shape = var_2408, x = var_2403)[name = string("q_35")]; tensor var_2410_cast_fp16 = mul(x = q_35, y = cos_f)[name = string("op_2410_cast_fp16")]; tensor var_2411_split_sizes_0 = const()[name = string("op_2411_split_sizes_0"), val = tensor([256, 256])]; int32 var_2411_axis_0 = const()[name = string("op_2411_axis_0"), val = int32(-1)]; tensor var_2411_0, tensor var_2411_1 = split(axis = var_2411_axis_0, split_sizes = var_2411_split_sizes_0, x = q_35)[name = string("op_2411")]; fp16 const_51_promoted = const()[name = string("const_51_promoted"), val = fp16(-0x1p+0)]; tensor var_2413 = mul(x = var_2411_1, y = const_51_promoted)[name = string("op_2413")]; int32 var_2415 = const()[name = string("op_2415"), val = int32(-1)]; bool var_2416_interleave_0 = const()[name = string("op_2416_interleave_0"), val = bool(false)]; tensor var_2416 = concat(axis = var_2415, interleave = var_2416_interleave_0, values = (var_2413, var_2411_0))[name = string("op_2416")]; tensor var_2417_cast_fp16 = mul(x = var_2416, y = sin_f)[name = string("op_2417_cast_fp16")]; tensor q_39_cast_fp16 = add(x = var_2410_cast_fp16, y = var_2417_cast_fp16)[name = string("q_39_cast_fp16")]; tensor linear_38_bias_0 = const()[name = string("linear_38_bias_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146772224)))]; tensor var_2422 = linear(bias = linear_38_bias_0, weight = layers_4_self_attn_k_proj_weight_palettized, x = var_2359_cast_fp16)[name = string("linear_38")]; tensor var_2427 = const()[name = string("op_2427"), val = tensor([1, 1, 512, 1])]; tensor var_2428 = reshape(shape = var_2427, x = var_2422)[name = string("op_2428")]; tensor var_2433 = const()[name = string("op_2433"), val = tensor([0, 1, 3, 2])]; tensor var_2442 = linear(bias = linear_38_bias_0, weight = layers_4_self_attn_v_proj_weight_palettized, x = var_2359_cast_fp16)[name = string("linear_39")]; tensor var_2447 = const()[name = string("op_2447"), val = tensor([1, 1, 512, 1])]; tensor var_2448 = reshape(shape = var_2447, x = var_2442)[name = string("op_2448")]; tensor var_2453 = const()[name = string("op_2453"), val = tensor([0, 1, 3, 2])]; tensor var_2463 = const()[name = string("op_2463"), val = tensor([1, 1, 512])]; tensor var_2434 = transpose(perm = var_2433, x = var_2428)[name = string("transpose_38")]; tensor x_83 = reshape(shape = var_2463, x = var_2434)[name = string("x_83")]; int32 var_2469 = const()[name = string("op_2469"), val = int32(-1)]; fp16 const_52_promoted = const()[name = string("const_52_promoted"), val = fp16(-0x1p+0)]; tensor var_2471 = mul(x = x_83, y = const_52_promoted)[name = string("op_2471")]; bool input_131_interleave_0 = const()[name = string("input_131_interleave_0"), val = bool(false)]; tensor input_131 = concat(axis = var_2469, interleave = input_131_interleave_0, values = (x_83, var_2471))[name = string("input_131")]; tensor normed_125_axes_0 = const()[name = string("normed_125_axes_0"), val = tensor([-1])]; fp16 var_2466_to_fp16 = const()[name = string("op_2466_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_125_cast_fp16 = layer_norm(axes = normed_125_axes_0, epsilon = var_2466_to_fp16, x = input_131)[name = string("normed_125_cast_fp16")]; tensor var_2476_split_sizes_0 = const()[name = string("op_2476_split_sizes_0"), val = tensor([512, 512])]; int32 var_2476_axis_0 = const()[name = string("op_2476_axis_0"), val = int32(-1)]; tensor var_2476_0, tensor var_2476_1 = split(axis = var_2476_axis_0, split_sizes = var_2476_split_sizes_0, x = normed_125_cast_fp16)[name = string("op_2476")]; tensor var_2478 = mul(x = var_2476_0, y = layers_4_self_attn_k_norm_weight)[name = string("op_2478")]; tensor var_2483 = const()[name = string("op_2483"), val = tensor([1, 1, 1, 512])]; tensor q_37 = reshape(shape = var_2483, x = var_2478)[name = string("q_37")]; fp16 var_2485_promoted = const()[name = string("op_2485_promoted"), val = fp16(0x1p+1)]; tensor var_2454 = transpose(perm = var_2453, x = var_2448)[name = string("transpose_37")]; tensor var_2486 = pow(x = var_2454, y = var_2485_promoted)[name = string("op_2486")]; tensor var_2491_axes_0 = const()[name = string("op_2491_axes_0"), val = tensor([-1])]; bool var_2491_keep_dims_0 = const()[name = string("op_2491_keep_dims_0"), val = bool(true)]; tensor var_2491 = reduce_mean(axes = var_2491_axes_0, keep_dims = var_2491_keep_dims_0, x = var_2486)[name = string("op_2491")]; fp16 var_2493_to_fp16 = const()[name = string("op_2493_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_9_cast_fp16 = add(x = var_2491, y = var_2493_to_fp16)[name = string("mean_sq_9_cast_fp16")]; fp32 var_2495_epsilon_0 = const()[name = string("op_2495_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_2495_cast_fp16 = rsqrt(epsilon = var_2495_epsilon_0, x = mean_sq_9_cast_fp16)[name = string("op_2495_cast_fp16")]; tensor v_cast_fp16 = mul(x = var_2454, y = var_2495_cast_fp16)[name = string("v_cast_fp16")]; tensor var_2497_cast_fp16 = mul(x = q_37, y = cos_f)[name = string("op_2497_cast_fp16")]; tensor var_2498_split_sizes_0 = const()[name = string("op_2498_split_sizes_0"), val = tensor([256, 256])]; int32 var_2498_axis_0 = const()[name = string("op_2498_axis_0"), val = int32(-1)]; tensor var_2498_0, tensor var_2498_1 = split(axis = var_2498_axis_0, split_sizes = var_2498_split_sizes_0, x = q_37)[name = string("op_2498")]; fp16 const_53_promoted = const()[name = string("const_53_promoted"), val = fp16(-0x1p+0)]; tensor var_2500 = mul(x = var_2498_1, y = const_53_promoted)[name = string("op_2500")]; int32 var_2502 = const()[name = string("op_2502"), val = int32(-1)]; bool var_2503_interleave_0 = const()[name = string("op_2503_interleave_0"), val = bool(false)]; tensor var_2503 = concat(axis = var_2502, interleave = var_2503_interleave_0, values = (var_2500, var_2498_0))[name = string("op_2503")]; tensor var_2504_cast_fp16 = mul(x = var_2503, y = sin_f)[name = string("op_2504_cast_fp16")]; tensor k_11_cast_fp16 = add(x = var_2497_cast_fp16, y = var_2504_cast_fp16)[name = string("k_11_cast_fp16")]; int32 var_2508 = const()[name = string("op_2508"), val = int32(1)]; tensor var_2509 = add(x = current_pos, y = var_2508)[name = string("op_2509")]; tensor read_state_1 = read_state(input = kv_cache_full)[name = string("read_state_1")]; tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([0])]; tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([1])]; int32 concat_50_axis_0 = const()[name = string("concat_50_axis_0"), val = int32(0)]; bool concat_50_interleave_0 = const()[name = string("concat_50_interleave_0"), val = bool(false)]; tensor concat_50 = concat(axis = concat_50_axis_0, interleave = concat_50_interleave_0, values = (expand_dims_48, expand_dims_49, current_pos, expand_dims_51))[name = string("concat_50")]; tensor concat_51_values1_0 = const()[name = string("concat_51_values1_0"), val = tensor([0])]; tensor concat_51_values3_0 = const()[name = string("concat_51_values3_0"), val = tensor([0])]; int32 concat_51_axis_0 = const()[name = string("concat_51_axis_0"), val = int32(0)]; bool concat_51_interleave_0 = const()[name = string("concat_51_interleave_0"), val = bool(false)]; tensor concat_51 = concat(axis = concat_51_axis_0, interleave = concat_51_interleave_0, values = (expand_dims_52, concat_51_values1_0, var_2509, concat_51_values3_0))[name = string("concat_51")]; tensor kv_cache_full_internal_tensor_assign_1_stride_0 = const()[name = string("kv_cache_full_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_full_internal_tensor_assign_1_begin_mask_0 = const()[name = string("kv_cache_full_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_full_internal_tensor_assign_1_end_mask_0 = const()[name = string("kv_cache_full_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_full_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("kv_cache_full_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_full_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_50, begin_mask = kv_cache_full_internal_tensor_assign_1_begin_mask_0, end = concat_51, end_mask = kv_cache_full_internal_tensor_assign_1_end_mask_0, squeeze_mask = kv_cache_full_internal_tensor_assign_1_squeeze_mask_0, stride = kv_cache_full_internal_tensor_assign_1_stride_0, update = k_11_cast_fp16, x = read_state_1)[name = string("kv_cache_full_internal_tensor_assign_1_cast_fp16")]; write_state(data = kv_cache_full_internal_tensor_assign_1_cast_fp16, input = kv_cache_full)[name = string("coreml_update_state_8_write_state")]; tensor coreml_update_state_24 = read_state(input = kv_cache_full)[name = string("coreml_update_state_8")]; tensor expand_dims_54 = const()[name = string("expand_dims_54"), val = tensor([1])]; tensor expand_dims_55 = const()[name = string("expand_dims_55"), val = tensor([0])]; tensor expand_dims_57 = const()[name = string("expand_dims_57"), val = tensor([0])]; tensor expand_dims_58 = const()[name = string("expand_dims_58"), val = tensor([2])]; int32 concat_54_axis_0 = const()[name = string("concat_54_axis_0"), val = int32(0)]; bool concat_54_interleave_0 = const()[name = string("concat_54_interleave_0"), val = bool(false)]; tensor concat_54 = concat(axis = concat_54_axis_0, interleave = concat_54_interleave_0, values = (expand_dims_54, expand_dims_55, current_pos, expand_dims_57))[name = string("concat_54")]; tensor concat_55_values1_0 = const()[name = string("concat_55_values1_0"), val = tensor([0])]; tensor concat_55_values3_0 = const()[name = string("concat_55_values3_0"), val = tensor([0])]; int32 concat_55_axis_0 = const()[name = string("concat_55_axis_0"), val = int32(0)]; bool concat_55_interleave_0 = const()[name = string("concat_55_interleave_0"), val = bool(false)]; tensor concat_55 = concat(axis = concat_55_axis_0, interleave = concat_55_interleave_0, values = (expand_dims_58, concat_55_values1_0, var_2509, concat_55_values3_0))[name = string("concat_55")]; tensor kv_cache_full_internal_tensor_assign_2_stride_0 = const()[name = string("kv_cache_full_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_full_internal_tensor_assign_2_begin_mask_0 = const()[name = string("kv_cache_full_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_full_internal_tensor_assign_2_end_mask_0 = const()[name = string("kv_cache_full_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_full_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("kv_cache_full_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_full_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_54, begin_mask = kv_cache_full_internal_tensor_assign_2_begin_mask_0, end = concat_55, end_mask = kv_cache_full_internal_tensor_assign_2_end_mask_0, squeeze_mask = kv_cache_full_internal_tensor_assign_2_squeeze_mask_0, stride = kv_cache_full_internal_tensor_assign_2_stride_0, update = v_cast_fp16, x = coreml_update_state_24)[name = string("kv_cache_full_internal_tensor_assign_2_cast_fp16")]; write_state(data = kv_cache_full_internal_tensor_assign_2_cast_fp16, input = kv_cache_full)[name = string("coreml_update_state_9_write_state")]; tensor coreml_update_state_25 = read_state(input = kv_cache_full)[name = string("coreml_update_state_9")]; tensor var_2559_begin_0 = const()[name = string("op_2559_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2559_end_0 = const()[name = string("op_2559_end_0"), val = tensor([1, 1, 2048, 512])]; tensor var_2559_end_mask_0 = const()[name = string("op_2559_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2559_cast_fp16 = slice_by_index(begin = var_2559_begin_0, end = var_2559_end_0, end_mask = var_2559_end_mask_0, x = coreml_update_state_25)[name = string("op_2559_cast_fp16")]; tensor var_2579_begin_0 = const()[name = string("op_2579_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_2579_end_0 = const()[name = string("op_2579_end_0"), val = tensor([1, 1, 2048, 512])]; tensor var_2579_end_mask_0 = const()[name = string("op_2579_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2579_cast_fp16 = slice_by_index(begin = var_2579_begin_0, end = var_2579_end_0, end_mask = var_2579_end_mask_0, x = coreml_update_state_25)[name = string("op_2579_cast_fp16")]; tensor transpose_16_perm_0 = const()[name = string("transpose_16_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_8_reps_0 = const()[name = string("tile_8_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_16_cast_fp16 = transpose(perm = transpose_16_perm_0, x = var_2559_cast_fp16)[name = string("transpose_36")]; tensor tile_8_cast_fp16 = tile(reps = tile_8_reps_0, x = transpose_16_cast_fp16)[name = string("tile_8_cast_fp16")]; tensor concat_56 = const()[name = string("concat_56"), val = tensor([8, 1, 1, 2048, 512])]; tensor reshape_16_cast_fp16 = reshape(shape = concat_56, x = tile_8_cast_fp16)[name = string("reshape_16_cast_fp16")]; tensor transpose_17_perm_0 = const()[name = string("transpose_17_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_57 = const()[name = string("concat_57"), val = tensor([-1, 1, 2048, 512])]; tensor transpose_17_cast_fp16 = transpose(perm = transpose_17_perm_0, x = reshape_16_cast_fp16)[name = string("transpose_35")]; tensor reshape_17_cast_fp16 = reshape(shape = concat_57, x = transpose_17_cast_fp16)[name = string("reshape_17_cast_fp16")]; tensor transpose_36_perm_0 = const()[name = string("transpose_36_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_18_perm_0 = const()[name = string("transpose_18_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_9_reps_0 = const()[name = string("tile_9_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_18_cast_fp16 = transpose(perm = transpose_18_perm_0, x = var_2579_cast_fp16)[name = string("transpose_34")]; tensor tile_9_cast_fp16 = tile(reps = tile_9_reps_0, x = transpose_18_cast_fp16)[name = string("tile_9_cast_fp16")]; tensor concat_58 = const()[name = string("concat_58"), val = tensor([8, 1, 1, 2048, 512])]; tensor reshape_18_cast_fp16 = reshape(shape = concat_58, x = tile_9_cast_fp16)[name = string("reshape_18_cast_fp16")]; tensor transpose_19_perm_0 = const()[name = string("transpose_19_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_59 = const()[name = string("concat_59"), val = tensor([-1, 1, 2048, 512])]; tensor transpose_19_cast_fp16 = transpose(perm = transpose_19_perm_0, x = reshape_18_cast_fp16)[name = string("transpose_33")]; tensor reshape_19_cast_fp16 = reshape(shape = concat_59, x = transpose_19_cast_fp16)[name = string("reshape_19_cast_fp16")]; tensor V_expanded_9_perm_0 = const()[name = string("V_expanded_9_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_17_transpose_x_0 = const()[name = string("attn_weights_17_transpose_x_0"), val = bool(false)]; bool attn_weights_17_transpose_y_0 = const()[name = string("attn_weights_17_transpose_y_0"), val = bool(false)]; tensor transpose_36_cast_fp16 = transpose(perm = transpose_36_perm_0, x = reshape_17_cast_fp16)[name = string("transpose_32")]; tensor attn_weights_17_cast_fp16 = matmul(transpose_x = attn_weights_17_transpose_x_0, transpose_y = attn_weights_17_transpose_y_0, x = q_39_cast_fp16, y = transpose_36_cast_fp16)[name = string("attn_weights_17_cast_fp16")]; tensor x_87_cast_fp16 = add(x = attn_weights_17_cast_fp16, y = causal_mask_full)[name = string("x_87_cast_fp16")]; tensor reduce_max_4_axes_0 = const()[name = string("reduce_max_4_axes_0"), val = tensor([-1])]; bool reduce_max_4_keep_dims_0 = const()[name = string("reduce_max_4_keep_dims_0"), val = bool(true)]; tensor reduce_max_4 = reduce_max(axes = reduce_max_4_axes_0, keep_dims = reduce_max_4_keep_dims_0, x = x_87_cast_fp16)[name = string("reduce_max_4")]; tensor var_2624 = sub(x = x_87_cast_fp16, y = reduce_max_4)[name = string("op_2624")]; tensor var_2630 = exp(x = var_2624)[name = string("op_2630")]; tensor var_2640_axes_0 = const()[name = string("op_2640_axes_0"), val = tensor([-1])]; bool var_2640_keep_dims_0 = const()[name = string("op_2640_keep_dims_0"), val = bool(true)]; tensor var_2640 = reduce_sum(axes = var_2640_axes_0, keep_dims = var_2640_keep_dims_0, x = var_2630)[name = string("op_2640")]; tensor var_2646_cast_fp16 = real_div(x = var_2630, y = var_2640)[name = string("op_2646_cast_fp16")]; bool attn_output_17_transpose_x_0 = const()[name = string("attn_output_17_transpose_x_0"), val = bool(false)]; bool attn_output_17_transpose_y_0 = const()[name = string("attn_output_17_transpose_y_0"), val = bool(false)]; tensor V_expanded_9_cast_fp16 = transpose(perm = V_expanded_9_perm_0, x = reshape_19_cast_fp16)[name = string("transpose_31")]; tensor attn_output_17_cast_fp16 = matmul(transpose_x = attn_output_17_transpose_x_0, transpose_y = attn_output_17_transpose_y_0, x = var_2646_cast_fp16, y = V_expanded_9_cast_fp16)[name = string("attn_output_17_cast_fp16")]; tensor var_2657 = const()[name = string("op_2657"), val = tensor([0, 2, 1, 3])]; tensor var_2664 = const()[name = string("op_2664"), val = tensor([1, 1, -1])]; tensor var_2658_cast_fp16 = transpose(perm = var_2657, x = attn_output_17_cast_fp16)[name = string("transpose_30")]; tensor input_133_cast_fp16 = reshape(shape = var_2664, x = var_2658_cast_fp16)[name = string("input_133_cast_fp16")]; tensor layers_4_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146773312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(149919104))))[name = string("layers_4_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_40_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_4_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_133_cast_fp16)[name = string("linear_40_cast_fp16")]; int32 var_2673 = const()[name = string("op_2673"), val = int32(-1)]; fp16 const_54_promoted_to_fp16 = const()[name = string("const_54_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2675_cast_fp16 = mul(x = linear_40_cast_fp16, y = const_54_promoted_to_fp16)[name = string("op_2675_cast_fp16")]; bool input_135_interleave_0 = const()[name = string("input_135_interleave_0"), val = bool(false)]; tensor input_135_cast_fp16 = concat(axis = var_2673, interleave = input_135_interleave_0, values = (linear_40_cast_fp16, var_2675_cast_fp16))[name = string("input_135_cast_fp16")]; tensor normed_129_axes_0 = const()[name = string("normed_129_axes_0"), val = tensor([-1])]; fp16 var_2670_to_fp16 = const()[name = string("op_2670_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_129_cast_fp16 = layer_norm(axes = normed_129_axes_0, epsilon = var_2670_to_fp16, x = input_135_cast_fp16)[name = string("normed_129_cast_fp16")]; tensor var_2680_split_sizes_0 = const()[name = string("op_2680_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2680_axis_0 = const()[name = string("op_2680_axis_0"), val = int32(-1)]; tensor var_2680_cast_fp16_0, tensor var_2680_cast_fp16_1 = split(axis = var_2680_axis_0, split_sizes = var_2680_split_sizes_0, x = normed_129_cast_fp16)[name = string("op_2680_cast_fp16")]; tensor layers_4_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_4_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(149920704)))]; tensor attn_output_19_cast_fp16 = mul(x = var_2680_cast_fp16_0, y = layers_4_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_19_cast_fp16")]; tensor x_93_cast_fp16 = add(x = x_79_cast_fp16, y = attn_output_19_cast_fp16)[name = string("x_93_cast_fp16")]; int32 var_2689 = const()[name = string("op_2689"), val = int32(-1)]; fp16 const_55_promoted_to_fp16 = const()[name = string("const_55_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2691_cast_fp16 = mul(x = x_93_cast_fp16, y = const_55_promoted_to_fp16)[name = string("op_2691_cast_fp16")]; bool input_137_interleave_0 = const()[name = string("input_137_interleave_0"), val = bool(false)]; tensor input_137_cast_fp16 = concat(axis = var_2689, interleave = input_137_interleave_0, values = (x_93_cast_fp16, var_2691_cast_fp16))[name = string("input_137_cast_fp16")]; tensor normed_133_axes_0 = const()[name = string("normed_133_axes_0"), val = tensor([-1])]; fp16 var_2686_to_fp16 = const()[name = string("op_2686_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_133_cast_fp16 = layer_norm(axes = normed_133_axes_0, epsilon = var_2686_to_fp16, x = input_137_cast_fp16)[name = string("normed_133_cast_fp16")]; tensor var_2696_split_sizes_0 = const()[name = string("op_2696_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2696_axis_0 = const()[name = string("op_2696_axis_0"), val = int32(-1)]; tensor var_2696_cast_fp16_0, tensor var_2696_cast_fp16_1 = split(axis = var_2696_axis_0, split_sizes = var_2696_split_sizes_0, x = normed_133_cast_fp16)[name = string("op_2696_cast_fp16")]; tensor layers_4_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_4_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(149923840)))]; tensor var_2698_cast_fp16 = mul(x = var_2696_cast_fp16_0, y = layers_4_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("op_2698_cast_fp16")]; tensor gate_17 = linear(bias = linear_5_bias_0, weight = layers_4_mlp_gate_proj_weight_palettized, x = var_2698_cast_fp16)[name = string("linear_41")]; tensor up_9 = linear(bias = linear_5_bias_0, weight = layers_4_mlp_up_proj_weight_palettized, x = var_2698_cast_fp16)[name = string("linear_42")]; string gate_19_mode_0 = const()[name = string("gate_19_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_19 = gelu(mode = gate_19_mode_0, x = gate_17)[name = string("gate_19")]; tensor input_141 = mul(x = gate_19, y = up_9)[name = string("input_141")]; tensor x_95 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_4_mlp_down_proj_weight_palettized, x = input_141)[name = string("linear_43")]; int32 var_2720 = const()[name = string("op_2720"), val = int32(-1)]; fp16 const_56_promoted = const()[name = string("const_56_promoted"), val = fp16(-0x1p+0)]; tensor var_2722 = mul(x = x_95, y = const_56_promoted)[name = string("op_2722")]; bool input_143_interleave_0 = const()[name = string("input_143_interleave_0"), val = bool(false)]; tensor input_143 = concat(axis = var_2720, interleave = input_143_interleave_0, values = (x_95, var_2722))[name = string("input_143")]; tensor normed_137_axes_0 = const()[name = string("normed_137_axes_0"), val = tensor([-1])]; fp16 var_2717_to_fp16 = const()[name = string("op_2717_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_137_cast_fp16 = layer_norm(axes = normed_137_axes_0, epsilon = var_2717_to_fp16, x = input_143)[name = string("normed_137_cast_fp16")]; tensor var_2727_split_sizes_0 = const()[name = string("op_2727_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2727_axis_0 = const()[name = string("op_2727_axis_0"), val = int32(-1)]; tensor var_2727_0, tensor var_2727_1 = split(axis = var_2727_axis_0, split_sizes = var_2727_split_sizes_0, x = normed_137_cast_fp16)[name = string("op_2727")]; tensor hidden_states_35 = mul(x = var_2727_0, y = layers_4_post_feedforward_layernorm_weight)[name = string("hidden_states_35")]; tensor hidden_states_37_cast_fp16 = add(x = x_93_cast_fp16, y = hidden_states_35)[name = string("hidden_states_37_cast_fp16")]; tensor per_layer_slice_9_begin_0 = const()[name = string("per_layer_slice_9_begin_0"), val = tensor([0, 0, 1024])]; tensor per_layer_slice_9_end_0 = const()[name = string("per_layer_slice_9_end_0"), val = tensor([1, 1, 1280])]; tensor per_layer_slice_9_end_mask_0 = const()[name = string("per_layer_slice_9_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_9_cast_fp16 = slice_by_index(begin = per_layer_slice_9_begin_0, end = per_layer_slice_9_end_0, end_mask = per_layer_slice_9_end_mask_0, x = per_layer_combined_out)[name = string("per_layer_slice_9_cast_fp16")]; tensor gated_17 = linear(bias = linear_2_bias_0, weight = layers_4_per_layer_input_gate_weight_palettized, x = hidden_states_37_cast_fp16)[name = string("linear_44")]; string gated_19_mode_0 = const()[name = string("gated_19_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_19 = gelu(mode = gated_19_mode_0, x = gated_17)[name = string("gated_19")]; tensor input_147_cast_fp16 = mul(x = gated_19, y = per_layer_slice_9_cast_fp16)[name = string("input_147_cast_fp16")]; tensor layers_4_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(149926976))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(150123648))))[name = string("layers_4_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_45_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_4_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_147_cast_fp16)[name = string("linear_45_cast_fp16")]; int32 var_2765 = const()[name = string("op_2765"), val = int32(-1)]; fp16 const_57_promoted_to_fp16 = const()[name = string("const_57_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2767_cast_fp16 = mul(x = linear_45_cast_fp16, y = const_57_promoted_to_fp16)[name = string("op_2767_cast_fp16")]; bool input_149_interleave_0 = const()[name = string("input_149_interleave_0"), val = bool(false)]; tensor input_149_cast_fp16 = concat(axis = var_2765, interleave = input_149_interleave_0, values = (linear_45_cast_fp16, var_2767_cast_fp16))[name = string("input_149_cast_fp16")]; tensor normed_141_axes_0 = const()[name = string("normed_141_axes_0"), val = tensor([-1])]; fp16 var_2762_to_fp16 = const()[name = string("op_2762_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_141_cast_fp16 = layer_norm(axes = normed_141_axes_0, epsilon = var_2762_to_fp16, x = input_149_cast_fp16)[name = string("normed_141_cast_fp16")]; tensor var_2772_split_sizes_0 = const()[name = string("op_2772_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2772_axis_0 = const()[name = string("op_2772_axis_0"), val = int32(-1)]; tensor var_2772_cast_fp16_0, tensor var_2772_cast_fp16_1 = split(axis = var_2772_axis_0, split_sizes = var_2772_split_sizes_0, x = normed_141_cast_fp16)[name = string("op_2772_cast_fp16")]; tensor layers_4_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_4_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(150125248)))]; tensor hidden_states_39_cast_fp16 = mul(x = var_2772_cast_fp16_0, y = layers_4_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_39_cast_fp16")]; tensor hidden_states_41_cast_fp16 = add(x = hidden_states_37_cast_fp16, y = hidden_states_39_cast_fp16)[name = string("hidden_states_41_cast_fp16")]; tensor const_58_promoted_to_fp16 = const()[name = string("const_58_promoted_to_fp16"), val = tensor([0x1.fep-2])]; tensor x_99_cast_fp16 = mul(x = hidden_states_41_cast_fp16, y = const_58_promoted_to_fp16)[name = string("x_99_cast_fp16")]; int32 var_2787 = const()[name = string("op_2787"), val = int32(-1)]; fp16 const_59_promoted_to_fp16 = const()[name = string("const_59_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2789_cast_fp16 = mul(x = x_99_cast_fp16, y = const_59_promoted_to_fp16)[name = string("op_2789_cast_fp16")]; bool input_151_interleave_0 = const()[name = string("input_151_interleave_0"), val = bool(false)]; tensor input_151_cast_fp16 = concat(axis = var_2787, interleave = input_151_interleave_0, values = (x_99_cast_fp16, var_2789_cast_fp16))[name = string("input_151_cast_fp16")]; tensor normed_145_axes_0 = const()[name = string("normed_145_axes_0"), val = tensor([-1])]; fp16 var_2784_to_fp16 = const()[name = string("op_2784_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_145_cast_fp16 = layer_norm(axes = normed_145_axes_0, epsilon = var_2784_to_fp16, x = input_151_cast_fp16)[name = string("normed_145_cast_fp16")]; tensor var_2794_split_sizes_0 = const()[name = string("op_2794_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2794_axis_0 = const()[name = string("op_2794_axis_0"), val = int32(-1)]; tensor var_2794_cast_fp16_0, tensor var_2794_cast_fp16_1 = split(axis = var_2794_axis_0, split_sizes = var_2794_split_sizes_0, x = normed_145_cast_fp16)[name = string("op_2794_cast_fp16")]; tensor layers_5_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_5_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(150128384)))]; tensor var_2796_cast_fp16 = mul(x = var_2794_cast_fp16_0, y = layers_5_input_layernorm_weight_promoted_to_fp16)[name = string("op_2796_cast_fp16")]; tensor var_2804 = linear(bias = linear_1_bias_0, weight = layers_5_self_attn_q_proj_weight_palettized, x = var_2796_cast_fp16)[name = string("linear_46")]; tensor var_2809 = const()[name = string("op_2809"), val = tensor([1, 8, 256, 1])]; tensor var_2810 = reshape(shape = var_2809, x = var_2804)[name = string("op_2810")]; tensor var_2815 = const()[name = string("op_2815"), val = tensor([0, 1, 3, 2])]; tensor var_2825 = const()[name = string("op_2825"), val = tensor([1, 8, 256])]; tensor var_2816 = transpose(perm = var_2815, x = var_2810)[name = string("transpose_29")]; tensor x_101 = reshape(shape = var_2825, x = var_2816)[name = string("x_101")]; int32 var_2831 = const()[name = string("op_2831"), val = int32(-1)]; fp16 const_60_promoted = const()[name = string("const_60_promoted"), val = fp16(-0x1p+0)]; tensor var_2833 = mul(x = x_101, y = const_60_promoted)[name = string("op_2833")]; bool input_155_interleave_0 = const()[name = string("input_155_interleave_0"), val = bool(false)]; tensor input_155 = concat(axis = var_2831, interleave = input_155_interleave_0, values = (x_101, var_2833))[name = string("input_155")]; tensor normed_149_axes_0 = const()[name = string("normed_149_axes_0"), val = tensor([-1])]; fp16 var_2828_to_fp16 = const()[name = string("op_2828_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_149_cast_fp16 = layer_norm(axes = normed_149_axes_0, epsilon = var_2828_to_fp16, x = input_155)[name = string("normed_149_cast_fp16")]; tensor var_2838_split_sizes_0 = const()[name = string("op_2838_split_sizes_0"), val = tensor([256, 256])]; int32 var_2838_axis_0 = const()[name = string("op_2838_axis_0"), val = int32(-1)]; tensor var_2838_0, tensor var_2838_1 = split(axis = var_2838_axis_0, split_sizes = var_2838_split_sizes_0, x = normed_149_cast_fp16)[name = string("op_2838")]; tensor var_2840 = mul(x = var_2838_0, y = layers_5_self_attn_q_norm_weight)[name = string("op_2840")]; tensor var_2845 = const()[name = string("op_2845"), val = tensor([1, 8, 1, 256])]; tensor q_43 = reshape(shape = var_2845, x = var_2840)[name = string("q_43")]; tensor var_2847_cast_fp16 = mul(x = q_43, y = cos_s)[name = string("op_2847_cast_fp16")]; tensor var_2848_split_sizes_0 = const()[name = string("op_2848_split_sizes_0"), val = tensor([128, 128])]; int32 var_2848_axis_0 = const()[name = string("op_2848_axis_0"), val = int32(-1)]; tensor var_2848_0, tensor var_2848_1 = split(axis = var_2848_axis_0, split_sizes = var_2848_split_sizes_0, x = q_43)[name = string("op_2848")]; fp16 const_61_promoted = const()[name = string("const_61_promoted"), val = fp16(-0x1p+0)]; tensor var_2850 = mul(x = var_2848_1, y = const_61_promoted)[name = string("op_2850")]; int32 var_2852 = const()[name = string("op_2852"), val = int32(-1)]; bool var_2853_interleave_0 = const()[name = string("op_2853_interleave_0"), val = bool(false)]; tensor var_2853 = concat(axis = var_2852, interleave = var_2853_interleave_0, values = (var_2850, var_2848_0))[name = string("op_2853")]; tensor var_2854_cast_fp16 = mul(x = var_2853, y = sin_s)[name = string("op_2854_cast_fp16")]; tensor q_47_cast_fp16 = add(x = var_2847_cast_fp16, y = var_2854_cast_fp16)[name = string("q_47_cast_fp16")]; tensor var_2859 = linear(bias = linear_2_bias_0, weight = layers_5_self_attn_k_proj_weight_palettized, x = var_2796_cast_fp16)[name = string("linear_47")]; tensor var_2864 = const()[name = string("op_2864"), val = tensor([1, 1, 256, 1])]; tensor var_2865 = reshape(shape = var_2864, x = var_2859)[name = string("op_2865")]; tensor var_2870 = const()[name = string("op_2870"), val = tensor([0, 1, 3, 2])]; tensor var_2879 = linear(bias = linear_2_bias_0, weight = layers_5_self_attn_v_proj_weight_palettized, x = var_2796_cast_fp16)[name = string("linear_48")]; tensor var_2884 = const()[name = string("op_2884"), val = tensor([1, 1, 256, 1])]; tensor var_2885 = reshape(shape = var_2884, x = var_2879)[name = string("op_2885")]; tensor var_2890 = const()[name = string("op_2890"), val = tensor([0, 1, 3, 2])]; tensor var_2900 = const()[name = string("op_2900"), val = tensor([1, 1, 256])]; tensor var_2871 = transpose(perm = var_2870, x = var_2865)[name = string("transpose_28")]; tensor x_103 = reshape(shape = var_2900, x = var_2871)[name = string("x_103")]; int32 var_2906 = const()[name = string("op_2906"), val = int32(-1)]; fp16 const_62_promoted = const()[name = string("const_62_promoted"), val = fp16(-0x1p+0)]; tensor var_2908 = mul(x = x_103, y = const_62_promoted)[name = string("op_2908")]; bool input_157_interleave_0 = const()[name = string("input_157_interleave_0"), val = bool(false)]; tensor input_157 = concat(axis = var_2906, interleave = input_157_interleave_0, values = (x_103, var_2908))[name = string("input_157")]; tensor normed_153_axes_0 = const()[name = string("normed_153_axes_0"), val = tensor([-1])]; fp16 var_2903_to_fp16 = const()[name = string("op_2903_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_153_cast_fp16 = layer_norm(axes = normed_153_axes_0, epsilon = var_2903_to_fp16, x = input_157)[name = string("normed_153_cast_fp16")]; tensor var_2913_split_sizes_0 = const()[name = string("op_2913_split_sizes_0"), val = tensor([256, 256])]; int32 var_2913_axis_0 = const()[name = string("op_2913_axis_0"), val = int32(-1)]; tensor var_2913_0, tensor var_2913_1 = split(axis = var_2913_axis_0, split_sizes = var_2913_split_sizes_0, x = normed_153_cast_fp16)[name = string("op_2913")]; tensor var_2915 = mul(x = var_2913_0, y = layers_0_self_attn_k_norm_weight)[name = string("op_2915")]; tensor var_2920 = const()[name = string("op_2920"), val = tensor([1, 1, 1, 256])]; tensor q_45 = reshape(shape = var_2920, x = var_2915)[name = string("q_45")]; fp16 var_2922_promoted = const()[name = string("op_2922_promoted"), val = fp16(0x1p+1)]; tensor var_2891 = transpose(perm = var_2890, x = var_2885)[name = string("transpose_27")]; tensor var_2923 = pow(x = var_2891, y = var_2922_promoted)[name = string("op_2923")]; tensor var_2928_axes_0 = const()[name = string("op_2928_axes_0"), val = tensor([-1])]; bool var_2928_keep_dims_0 = const()[name = string("op_2928_keep_dims_0"), val = bool(true)]; tensor var_2928 = reduce_mean(axes = var_2928_axes_0, keep_dims = var_2928_keep_dims_0, x = var_2923)[name = string("op_2928")]; fp16 var_2930_to_fp16 = const()[name = string("op_2930_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_11_cast_fp16 = add(x = var_2928, y = var_2930_to_fp16)[name = string("mean_sq_11_cast_fp16")]; fp32 var_2932_epsilon_0 = const()[name = string("op_2932_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_2932_cast_fp16 = rsqrt(epsilon = var_2932_epsilon_0, x = mean_sq_11_cast_fp16)[name = string("op_2932_cast_fp16")]; tensor input_161_cast_fp16 = mul(x = var_2891, y = var_2932_cast_fp16)[name = string("input_161_cast_fp16")]; tensor var_2934_cast_fp16 = mul(x = q_45, y = cos_s)[name = string("op_2934_cast_fp16")]; tensor var_2935_split_sizes_0 = const()[name = string("op_2935_split_sizes_0"), val = tensor([128, 128])]; int32 var_2935_axis_0 = const()[name = string("op_2935_axis_0"), val = int32(-1)]; tensor var_2935_0, tensor var_2935_1 = split(axis = var_2935_axis_0, split_sizes = var_2935_split_sizes_0, x = q_45)[name = string("op_2935")]; fp16 const_63_promoted = const()[name = string("const_63_promoted"), val = fp16(-0x1p+0)]; tensor var_2937 = mul(x = var_2935_1, y = const_63_promoted)[name = string("op_2937")]; int32 var_2939 = const()[name = string("op_2939"), val = int32(-1)]; bool var_2940_interleave_0 = const()[name = string("op_2940_interleave_0"), val = bool(false)]; tensor var_2940 = concat(axis = var_2939, interleave = var_2940_interleave_0, values = (var_2937, var_2935_0))[name = string("op_2940")]; tensor var_2941_cast_fp16 = mul(x = var_2940, y = sin_s)[name = string("op_2941_cast_fp16")]; tensor input_159_cast_fp16 = add(x = var_2934_cast_fp16, y = var_2941_cast_fp16)[name = string("input_159_cast_fp16")]; tensor k_padded_9_pad_0 = const()[name = string("k_padded_9_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_9_mode_0 = const()[name = string("k_padded_9_mode_0"), val = string("constant")]; fp16 const_64_to_fp16 = const()[name = string("const_64_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_9_cast_fp16 = pad(constant_val = const_64_to_fp16, mode = k_padded_9_mode_0, pad = k_padded_9_pad_0, x = input_159_cast_fp16)[name = string("k_padded_9_cast_fp16")]; tensor v_padded_9_pad_0 = const()[name = string("v_padded_9_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_9_mode_0 = const()[name = string("v_padded_9_mode_0"), val = string("constant")]; fp16 const_65_to_fp16 = const()[name = string("const_65_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_9_cast_fp16 = pad(constant_val = const_65_to_fp16, mode = v_padded_9_mode_0, pad = v_padded_9_pad_0, x = input_161_cast_fp16)[name = string("v_padded_9_cast_fp16")]; tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([8])]; tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([9])]; int32 concat_62_axis_0 = const()[name = string("concat_62_axis_0"), val = int32(0)]; bool concat_62_interleave_0 = const()[name = string("concat_62_interleave_0"), val = bool(false)]; tensor concat_62 = concat(axis = concat_62_axis_0, interleave = concat_62_interleave_0, values = (expand_dims_60, expand_dims_61, ring_pos, expand_dims_63))[name = string("concat_62")]; tensor concat_63_values1_0 = const()[name = string("concat_63_values1_0"), val = tensor([0])]; tensor concat_63_values3_0 = const()[name = string("concat_63_values3_0"), val = tensor([0])]; int32 concat_63_axis_0 = const()[name = string("concat_63_axis_0"), val = int32(0)]; bool concat_63_interleave_0 = const()[name = string("concat_63_interleave_0"), val = bool(false)]; tensor concat_63 = concat(axis = concat_63_axis_0, interleave = concat_63_interleave_0, values = (expand_dims_64, concat_63_values1_0, var_725, concat_63_values3_0))[name = string("concat_63")]; tensor kv_cache_sliding_internal_tensor_assign_9_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_9_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_9_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_62, begin_mask = kv_cache_sliding_internal_tensor_assign_9_begin_mask_0, end = concat_63, end_mask = kv_cache_sliding_internal_tensor_assign_9_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_9_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_9_stride_0, update = k_padded_9_cast_fp16, x = coreml_update_state_23)[name = string("kv_cache_sliding_internal_tensor_assign_9_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_9_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_10_write_state")]; tensor coreml_update_state_26 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_10")]; tensor expand_dims_66 = const()[name = string("expand_dims_66"), val = tensor([9])]; tensor expand_dims_67 = const()[name = string("expand_dims_67"), val = tensor([0])]; tensor expand_dims_69 = const()[name = string("expand_dims_69"), val = tensor([0])]; tensor expand_dims_70 = const()[name = string("expand_dims_70"), val = tensor([10])]; int32 concat_66_axis_0 = const()[name = string("concat_66_axis_0"), val = int32(0)]; bool concat_66_interleave_0 = const()[name = string("concat_66_interleave_0"), val = bool(false)]; tensor concat_66 = concat(axis = concat_66_axis_0, interleave = concat_66_interleave_0, values = (expand_dims_66, expand_dims_67, ring_pos, expand_dims_69))[name = string("concat_66")]; tensor concat_67_values1_0 = const()[name = string("concat_67_values1_0"), val = tensor([0])]; tensor concat_67_values3_0 = const()[name = string("concat_67_values3_0"), val = tensor([0])]; int32 concat_67_axis_0 = const()[name = string("concat_67_axis_0"), val = int32(0)]; bool concat_67_interleave_0 = const()[name = string("concat_67_interleave_0"), val = bool(false)]; tensor concat_67 = concat(axis = concat_67_axis_0, interleave = concat_67_interleave_0, values = (expand_dims_70, concat_67_values1_0, var_725, concat_67_values3_0))[name = string("concat_67")]; tensor kv_cache_sliding_internal_tensor_assign_10_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_10_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_10_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_66, begin_mask = kv_cache_sliding_internal_tensor_assign_10_begin_mask_0, end = concat_67, end_mask = kv_cache_sliding_internal_tensor_assign_10_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_10_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_10_stride_0, update = v_padded_9_cast_fp16, x = coreml_update_state_26)[name = string("kv_cache_sliding_internal_tensor_assign_10_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_10_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_11_write_state")]; tensor coreml_update_state_27 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_11")]; tensor var_3008_begin_0 = const()[name = string("op_3008_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3008_end_0 = const()[name = string("op_3008_end_0"), val = tensor([9, 1, 512, 512])]; tensor var_3008_end_mask_0 = const()[name = string("op_3008_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3008_cast_fp16 = slice_by_index(begin = var_3008_begin_0, end = var_3008_end_0, end_mask = var_3008_end_mask_0, x = coreml_update_state_27)[name = string("op_3008_cast_fp16")]; tensor K_sliding_slice_9_begin_0 = const()[name = string("K_sliding_slice_9_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_sliding_slice_9_end_0 = const()[name = string("K_sliding_slice_9_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_sliding_slice_9_end_mask_0 = const()[name = string("K_sliding_slice_9_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_sliding_slice_9_cast_fp16 = slice_by_index(begin = K_sliding_slice_9_begin_0, end = K_sliding_slice_9_end_0, end_mask = K_sliding_slice_9_end_mask_0, x = var_3008_cast_fp16)[name = string("K_sliding_slice_9_cast_fp16")]; tensor var_3028_begin_0 = const()[name = string("op_3028_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3028_end_0 = const()[name = string("op_3028_end_0"), val = tensor([10, 1, 512, 512])]; tensor var_3028_end_mask_0 = const()[name = string("op_3028_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3028_cast_fp16 = slice_by_index(begin = var_3028_begin_0, end = var_3028_end_0, end_mask = var_3028_end_mask_0, x = coreml_update_state_27)[name = string("op_3028_cast_fp16")]; tensor V_for_attn_9_begin_0 = const()[name = string("V_for_attn_9_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_9_end_0 = const()[name = string("V_for_attn_9_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_9_end_mask_0 = const()[name = string("V_for_attn_9_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_9_cast_fp16 = slice_by_index(begin = V_for_attn_9_begin_0, end = V_for_attn_9_end_0, end_mask = V_for_attn_9_end_mask_0, x = var_3028_cast_fp16)[name = string("V_for_attn_9_cast_fp16")]; tensor transpose_20_perm_0 = const()[name = string("transpose_20_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_10_reps_0 = const()[name = string("tile_10_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_20_cast_fp16 = transpose(perm = transpose_20_perm_0, x = K_sliding_slice_9_cast_fp16)[name = string("transpose_26")]; tensor tile_10_cast_fp16 = tile(reps = tile_10_reps_0, x = transpose_20_cast_fp16)[name = string("tile_10_cast_fp16")]; tensor concat_68 = const()[name = string("concat_68"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_20_cast_fp16 = reshape(shape = concat_68, x = tile_10_cast_fp16)[name = string("reshape_20_cast_fp16")]; tensor transpose_21_perm_0 = const()[name = string("transpose_21_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_69 = const()[name = string("concat_69"), val = tensor([-1, 1, 512, 256])]; tensor transpose_21_cast_fp16 = transpose(perm = transpose_21_perm_0, x = reshape_20_cast_fp16)[name = string("transpose_25")]; tensor reshape_21_cast_fp16 = reshape(shape = concat_69, x = transpose_21_cast_fp16)[name = string("reshape_21_cast_fp16")]; tensor transpose_37_perm_0 = const()[name = string("transpose_37_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_22_perm_0 = const()[name = string("transpose_22_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_11_reps_0 = const()[name = string("tile_11_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_22_cast_fp16 = transpose(perm = transpose_22_perm_0, x = V_for_attn_9_cast_fp16)[name = string("transpose_24")]; tensor tile_11_cast_fp16 = tile(reps = tile_11_reps_0, x = transpose_22_cast_fp16)[name = string("tile_11_cast_fp16")]; tensor concat_70 = const()[name = string("concat_70"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_22_cast_fp16 = reshape(shape = concat_70, x = tile_11_cast_fp16)[name = string("reshape_22_cast_fp16")]; tensor transpose_23_perm_0 = const()[name = string("transpose_23_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_71 = const()[name = string("concat_71"), val = tensor([-1, 1, 512, 256])]; tensor transpose_23_cast_fp16 = transpose(perm = transpose_23_perm_0, x = reshape_22_cast_fp16)[name = string("transpose_23")]; tensor reshape_23_cast_fp16 = reshape(shape = concat_71, x = transpose_23_cast_fp16)[name = string("reshape_23_cast_fp16")]; tensor V_expanded_11_perm_0 = const()[name = string("V_expanded_11_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_21_transpose_x_0 = const()[name = string("attn_weights_21_transpose_x_0"), val = bool(false)]; bool attn_weights_21_transpose_y_0 = const()[name = string("attn_weights_21_transpose_y_0"), val = bool(false)]; tensor transpose_37_cast_fp16 = transpose(perm = transpose_37_perm_0, x = reshape_21_cast_fp16)[name = string("transpose_22")]; tensor attn_weights_21_cast_fp16 = matmul(transpose_x = attn_weights_21_transpose_x_0, transpose_y = attn_weights_21_transpose_y_0, x = q_47_cast_fp16, y = transpose_37_cast_fp16)[name = string("attn_weights_21_cast_fp16")]; tensor x_107_cast_fp16 = add(x = attn_weights_21_cast_fp16, y = causal_mask_sliding)[name = string("x_107_cast_fp16")]; tensor reduce_max_5_axes_0 = const()[name = string("reduce_max_5_axes_0"), val = tensor([-1])]; bool reduce_max_5_keep_dims_0 = const()[name = string("reduce_max_5_keep_dims_0"), val = bool(true)]; tensor reduce_max_5 = reduce_max(axes = reduce_max_5_axes_0, keep_dims = reduce_max_5_keep_dims_0, x = x_107_cast_fp16)[name = string("reduce_max_5")]; tensor var_3073 = sub(x = x_107_cast_fp16, y = reduce_max_5)[name = string("op_3073")]; tensor var_3079 = exp(x = var_3073)[name = string("op_3079")]; tensor var_3089_axes_0 = const()[name = string("op_3089_axes_0"), val = tensor([-1])]; bool var_3089_keep_dims_0 = const()[name = string("op_3089_keep_dims_0"), val = bool(true)]; tensor var_3089 = reduce_sum(axes = var_3089_axes_0, keep_dims = var_3089_keep_dims_0, x = var_3079)[name = string("op_3089")]; tensor var_3095_cast_fp16 = real_div(x = var_3079, y = var_3089)[name = string("op_3095_cast_fp16")]; bool attn_output_21_transpose_x_0 = const()[name = string("attn_output_21_transpose_x_0"), val = bool(false)]; bool attn_output_21_transpose_y_0 = const()[name = string("attn_output_21_transpose_y_0"), val = bool(false)]; tensor V_expanded_11_cast_fp16 = transpose(perm = V_expanded_11_perm_0, x = reshape_23_cast_fp16)[name = string("transpose_21")]; tensor attn_output_21_cast_fp16 = matmul(transpose_x = attn_output_21_transpose_x_0, transpose_y = attn_output_21_transpose_y_0, x = var_3095_cast_fp16, y = V_expanded_11_cast_fp16)[name = string("attn_output_21_cast_fp16")]; tensor var_3106 = const()[name = string("op_3106"), val = tensor([0, 2, 1, 3])]; tensor var_3113 = const()[name = string("op_3113"), val = tensor([1, 1, -1])]; tensor var_3107_cast_fp16 = transpose(perm = var_3106, x = attn_output_21_cast_fp16)[name = string("transpose_20")]; tensor input_163_cast_fp16 = reshape(shape = var_3113, x = var_3107_cast_fp16)[name = string("input_163_cast_fp16")]; tensor layers_5_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(150131520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151704448))))[name = string("layers_5_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_49_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_5_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_163_cast_fp16)[name = string("linear_49_cast_fp16")]; int32 var_3122 = const()[name = string("op_3122"), val = int32(-1)]; fp16 const_66_promoted_to_fp16 = const()[name = string("const_66_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3124_cast_fp16 = mul(x = linear_49_cast_fp16, y = const_66_promoted_to_fp16)[name = string("op_3124_cast_fp16")]; bool input_165_interleave_0 = const()[name = string("input_165_interleave_0"), val = bool(false)]; tensor input_165_cast_fp16 = concat(axis = var_3122, interleave = input_165_interleave_0, values = (linear_49_cast_fp16, var_3124_cast_fp16))[name = string("input_165_cast_fp16")]; tensor normed_157_axes_0 = const()[name = string("normed_157_axes_0"), val = tensor([-1])]; fp16 var_3119_to_fp16 = const()[name = string("op_3119_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_157_cast_fp16 = layer_norm(axes = normed_157_axes_0, epsilon = var_3119_to_fp16, x = input_165_cast_fp16)[name = string("normed_157_cast_fp16")]; tensor var_3129_split_sizes_0 = const()[name = string("op_3129_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3129_axis_0 = const()[name = string("op_3129_axis_0"), val = int32(-1)]; tensor var_3129_cast_fp16_0, tensor var_3129_cast_fp16_1 = split(axis = var_3129_axis_0, split_sizes = var_3129_split_sizes_0, x = normed_157_cast_fp16)[name = string("op_3129_cast_fp16")]; tensor layers_5_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_5_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151706048)))]; tensor attn_output_23_cast_fp16 = mul(x = var_3129_cast_fp16_0, y = layers_5_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_23_cast_fp16")]; tensor x_113_cast_fp16 = add(x = x_99_cast_fp16, y = attn_output_23_cast_fp16)[name = string("x_113_cast_fp16")]; int32 var_3138 = const()[name = string("op_3138"), val = int32(-1)]; fp16 const_67_promoted_to_fp16 = const()[name = string("const_67_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3140_cast_fp16 = mul(x = x_113_cast_fp16, y = const_67_promoted_to_fp16)[name = string("op_3140_cast_fp16")]; bool input_167_interleave_0 = const()[name = string("input_167_interleave_0"), val = bool(false)]; tensor input_167_cast_fp16 = concat(axis = var_3138, interleave = input_167_interleave_0, values = (x_113_cast_fp16, var_3140_cast_fp16))[name = string("input_167_cast_fp16")]; tensor normed_161_axes_0 = const()[name = string("normed_161_axes_0"), val = tensor([-1])]; fp16 var_3135_to_fp16 = const()[name = string("op_3135_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_161_cast_fp16 = layer_norm(axes = normed_161_axes_0, epsilon = var_3135_to_fp16, x = input_167_cast_fp16)[name = string("normed_161_cast_fp16")]; tensor var_3145_split_sizes_0 = const()[name = string("op_3145_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3145_axis_0 = const()[name = string("op_3145_axis_0"), val = int32(-1)]; tensor var_3145_cast_fp16_0, tensor var_3145_cast_fp16_1 = split(axis = var_3145_axis_0, split_sizes = var_3145_split_sizes_0, x = normed_161_cast_fp16)[name = string("op_3145_cast_fp16")]; tensor layers_5_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_5_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151709184)))]; tensor var_3147_cast_fp16 = mul(x = var_3145_cast_fp16_0, y = layers_5_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("op_3147_cast_fp16")]; tensor gate_21 = linear(bias = linear_5_bias_0, weight = layers_5_mlp_gate_proj_weight_palettized, x = var_3147_cast_fp16)[name = string("linear_50")]; tensor up_11 = linear(bias = linear_5_bias_0, weight = layers_5_mlp_up_proj_weight_palettized, x = var_3147_cast_fp16)[name = string("linear_51")]; string gate_23_mode_0 = const()[name = string("gate_23_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_23 = gelu(mode = gate_23_mode_0, x = gate_21)[name = string("gate_23")]; tensor input_171 = mul(x = gate_23, y = up_11)[name = string("input_171")]; tensor x_115 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_5_mlp_down_proj_weight_palettized, x = input_171)[name = string("linear_52")]; int32 var_3169 = const()[name = string("op_3169"), val = int32(-1)]; fp16 const_68_promoted = const()[name = string("const_68_promoted"), val = fp16(-0x1p+0)]; tensor var_3171 = mul(x = x_115, y = const_68_promoted)[name = string("op_3171")]; bool input_173_interleave_0 = const()[name = string("input_173_interleave_0"), val = bool(false)]; tensor input_173 = concat(axis = var_3169, interleave = input_173_interleave_0, values = (x_115, var_3171))[name = string("input_173")]; tensor normed_165_axes_0 = const()[name = string("normed_165_axes_0"), val = tensor([-1])]; fp16 var_3166_to_fp16 = const()[name = string("op_3166_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_165_cast_fp16 = layer_norm(axes = normed_165_axes_0, epsilon = var_3166_to_fp16, x = input_173)[name = string("normed_165_cast_fp16")]; tensor var_3176_split_sizes_0 = const()[name = string("op_3176_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3176_axis_0 = const()[name = string("op_3176_axis_0"), val = int32(-1)]; tensor var_3176_0, tensor var_3176_1 = split(axis = var_3176_axis_0, split_sizes = var_3176_split_sizes_0, x = normed_165_cast_fp16)[name = string("op_3176")]; tensor hidden_states_43 = mul(x = var_3176_0, y = layers_5_post_feedforward_layernorm_weight)[name = string("hidden_states_43")]; tensor hidden_states_45_cast_fp16 = add(x = x_113_cast_fp16, y = hidden_states_43)[name = string("hidden_states_45_cast_fp16")]; tensor per_layer_slice_11_begin_0 = const()[name = string("per_layer_slice_11_begin_0"), val = tensor([0, 0, 1280])]; tensor per_layer_slice_11_end_0 = const()[name = string("per_layer_slice_11_end_0"), val = tensor([1, 1, 1536])]; tensor per_layer_slice_11_end_mask_0 = const()[name = string("per_layer_slice_11_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_11_cast_fp16 = slice_by_index(begin = per_layer_slice_11_begin_0, end = per_layer_slice_11_end_0, end_mask = per_layer_slice_11_end_mask_0, x = per_layer_combined_out)[name = string("per_layer_slice_11_cast_fp16")]; tensor gated_21 = linear(bias = linear_2_bias_0, weight = layers_5_per_layer_input_gate_weight_palettized, x = hidden_states_45_cast_fp16)[name = string("linear_53")]; string gated_23_mode_0 = const()[name = string("gated_23_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_23 = gelu(mode = gated_23_mode_0, x = gated_21)[name = string("gated_23")]; tensor input_177_cast_fp16 = mul(x = gated_23, y = per_layer_slice_11_cast_fp16)[name = string("input_177_cast_fp16")]; tensor layers_5_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151712320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151908992))))[name = string("layers_5_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_54_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_5_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_177_cast_fp16)[name = string("linear_54_cast_fp16")]; int32 var_3214 = const()[name = string("op_3214"), val = int32(-1)]; fp16 const_69_promoted_to_fp16 = const()[name = string("const_69_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3216_cast_fp16 = mul(x = linear_54_cast_fp16, y = const_69_promoted_to_fp16)[name = string("op_3216_cast_fp16")]; bool input_179_interleave_0 = const()[name = string("input_179_interleave_0"), val = bool(false)]; tensor input_179_cast_fp16 = concat(axis = var_3214, interleave = input_179_interleave_0, values = (linear_54_cast_fp16, var_3216_cast_fp16))[name = string("input_179_cast_fp16")]; tensor normed_169_axes_0 = const()[name = string("normed_169_axes_0"), val = tensor([-1])]; fp16 var_3211_to_fp16 = const()[name = string("op_3211_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_169_cast_fp16 = layer_norm(axes = normed_169_axes_0, epsilon = var_3211_to_fp16, x = input_179_cast_fp16)[name = string("normed_169_cast_fp16")]; tensor var_3221_split_sizes_0 = const()[name = string("op_3221_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3221_axis_0 = const()[name = string("op_3221_axis_0"), val = int32(-1)]; tensor var_3221_cast_fp16_0, tensor var_3221_cast_fp16_1 = split(axis = var_3221_axis_0, split_sizes = var_3221_split_sizes_0, x = normed_169_cast_fp16)[name = string("op_3221_cast_fp16")]; tensor layers_5_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_5_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151910592)))]; tensor hidden_states_47_cast_fp16 = mul(x = var_3221_cast_fp16_0, y = layers_5_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_47_cast_fp16")]; tensor hidden_states_49_cast_fp16 = add(x = hidden_states_45_cast_fp16, y = hidden_states_47_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; tensor const_70_promoted_to_fp16 = const()[name = string("const_70_promoted_to_fp16"), val = tensor([0x1.46p-1])]; tensor x_119_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_70_promoted_to_fp16)[name = string("x_119_cast_fp16")]; int32 var_3236 = const()[name = string("op_3236"), val = int32(-1)]; fp16 const_71_promoted_to_fp16 = const()[name = string("const_71_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3238_cast_fp16 = mul(x = x_119_cast_fp16, y = const_71_promoted_to_fp16)[name = string("op_3238_cast_fp16")]; bool input_181_interleave_0 = const()[name = string("input_181_interleave_0"), val = bool(false)]; tensor input_181_cast_fp16 = concat(axis = var_3236, interleave = input_181_interleave_0, values = (x_119_cast_fp16, var_3238_cast_fp16))[name = string("input_181_cast_fp16")]; tensor normed_173_axes_0 = const()[name = string("normed_173_axes_0"), val = tensor([-1])]; fp16 var_3233_to_fp16 = const()[name = string("op_3233_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_173_cast_fp16 = layer_norm(axes = normed_173_axes_0, epsilon = var_3233_to_fp16, x = input_181_cast_fp16)[name = string("normed_173_cast_fp16")]; tensor var_3243_split_sizes_0 = const()[name = string("op_3243_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3243_axis_0 = const()[name = string("op_3243_axis_0"), val = int32(-1)]; tensor var_3243_cast_fp16_0, tensor var_3243_cast_fp16_1 = split(axis = var_3243_axis_0, split_sizes = var_3243_split_sizes_0, x = normed_173_cast_fp16)[name = string("op_3243_cast_fp16")]; tensor layers_6_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_6_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151913728)))]; tensor var_3245_cast_fp16 = mul(x = var_3243_cast_fp16_0, y = layers_6_input_layernorm_weight_promoted_to_fp16)[name = string("op_3245_cast_fp16")]; tensor var_3253 = linear(bias = linear_1_bias_0, weight = layers_6_self_attn_q_proj_weight_palettized, x = var_3245_cast_fp16)[name = string("linear_55")]; tensor var_3258 = const()[name = string("op_3258"), val = tensor([1, 8, 256, 1])]; tensor var_3259 = reshape(shape = var_3258, x = var_3253)[name = string("op_3259")]; tensor var_3264 = const()[name = string("op_3264"), val = tensor([0, 1, 3, 2])]; tensor var_3274 = const()[name = string("op_3274"), val = tensor([1, 8, 256])]; tensor var_3265 = transpose(perm = var_3264, x = var_3259)[name = string("transpose_19")]; tensor x_121 = reshape(shape = var_3274, x = var_3265)[name = string("x_121")]; int32 var_3280 = const()[name = string("op_3280"), val = int32(-1)]; fp16 const_72_promoted = const()[name = string("const_72_promoted"), val = fp16(-0x1p+0)]; tensor var_3282 = mul(x = x_121, y = const_72_promoted)[name = string("op_3282")]; bool input_185_interleave_0 = const()[name = string("input_185_interleave_0"), val = bool(false)]; tensor input_185 = concat(axis = var_3280, interleave = input_185_interleave_0, values = (x_121, var_3282))[name = string("input_185")]; tensor normed_177_axes_0 = const()[name = string("normed_177_axes_0"), val = tensor([-1])]; fp16 var_3277_to_fp16 = const()[name = string("op_3277_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_177_cast_fp16 = layer_norm(axes = normed_177_axes_0, epsilon = var_3277_to_fp16, x = input_185)[name = string("normed_177_cast_fp16")]; tensor var_3287_split_sizes_0 = const()[name = string("op_3287_split_sizes_0"), val = tensor([256, 256])]; int32 var_3287_axis_0 = const()[name = string("op_3287_axis_0"), val = int32(-1)]; tensor var_3287_0, tensor var_3287_1 = split(axis = var_3287_axis_0, split_sizes = var_3287_split_sizes_0, x = normed_177_cast_fp16)[name = string("op_3287")]; tensor var_3289 = mul(x = var_3287_0, y = layers_1_self_attn_q_norm_weight)[name = string("op_3289")]; tensor var_3294 = const()[name = string("op_3294"), val = tensor([1, 8, 1, 256])]; tensor q_51 = reshape(shape = var_3294, x = var_3289)[name = string("q_51")]; tensor var_3296_cast_fp16 = mul(x = q_51, y = cos_s)[name = string("op_3296_cast_fp16")]; tensor var_3297_split_sizes_0 = const()[name = string("op_3297_split_sizes_0"), val = tensor([128, 128])]; int32 var_3297_axis_0 = const()[name = string("op_3297_axis_0"), val = int32(-1)]; tensor var_3297_0, tensor var_3297_1 = split(axis = var_3297_axis_0, split_sizes = var_3297_split_sizes_0, x = q_51)[name = string("op_3297")]; fp16 const_73_promoted = const()[name = string("const_73_promoted"), val = fp16(-0x1p+0)]; tensor var_3299 = mul(x = var_3297_1, y = const_73_promoted)[name = string("op_3299")]; int32 var_3301 = const()[name = string("op_3301"), val = int32(-1)]; bool var_3302_interleave_0 = const()[name = string("op_3302_interleave_0"), val = bool(false)]; tensor var_3302 = concat(axis = var_3301, interleave = var_3302_interleave_0, values = (var_3299, var_3297_0))[name = string("op_3302")]; tensor var_3303_cast_fp16 = mul(x = var_3302, y = sin_s)[name = string("op_3303_cast_fp16")]; tensor q_55_cast_fp16 = add(x = var_3296_cast_fp16, y = var_3303_cast_fp16)[name = string("q_55_cast_fp16")]; tensor var_3308 = linear(bias = linear_2_bias_0, weight = layers_6_self_attn_k_proj_weight_palettized, x = var_3245_cast_fp16)[name = string("linear_56")]; tensor var_3313 = const()[name = string("op_3313"), val = tensor([1, 1, 256, 1])]; tensor var_3314 = reshape(shape = var_3313, x = var_3308)[name = string("op_3314")]; tensor var_3319 = const()[name = string("op_3319"), val = tensor([0, 1, 3, 2])]; tensor var_3328 = linear(bias = linear_2_bias_0, weight = layers_6_self_attn_v_proj_weight_palettized, x = var_3245_cast_fp16)[name = string("linear_57")]; tensor var_3333 = const()[name = string("op_3333"), val = tensor([1, 1, 256, 1])]; tensor var_3334 = reshape(shape = var_3333, x = var_3328)[name = string("op_3334")]; tensor var_3339 = const()[name = string("op_3339"), val = tensor([0, 1, 3, 2])]; tensor var_3349 = const()[name = string("op_3349"), val = tensor([1, 1, 256])]; tensor var_3320 = transpose(perm = var_3319, x = var_3314)[name = string("transpose_18")]; tensor x_123 = reshape(shape = var_3349, x = var_3320)[name = string("x_123")]; int32 var_3355 = const()[name = string("op_3355"), val = int32(-1)]; fp16 const_74_promoted = const()[name = string("const_74_promoted"), val = fp16(-0x1p+0)]; tensor var_3357 = mul(x = x_123, y = const_74_promoted)[name = string("op_3357")]; bool input_187_interleave_0 = const()[name = string("input_187_interleave_0"), val = bool(false)]; tensor input_187 = concat(axis = var_3355, interleave = input_187_interleave_0, values = (x_123, var_3357))[name = string("input_187")]; tensor normed_181_axes_0 = const()[name = string("normed_181_axes_0"), val = tensor([-1])]; fp16 var_3352_to_fp16 = const()[name = string("op_3352_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_181_cast_fp16 = layer_norm(axes = normed_181_axes_0, epsilon = var_3352_to_fp16, x = input_187)[name = string("normed_181_cast_fp16")]; tensor var_3362_split_sizes_0 = const()[name = string("op_3362_split_sizes_0"), val = tensor([256, 256])]; int32 var_3362_axis_0 = const()[name = string("op_3362_axis_0"), val = int32(-1)]; tensor var_3362_0, tensor var_3362_1 = split(axis = var_3362_axis_0, split_sizes = var_3362_split_sizes_0, x = normed_181_cast_fp16)[name = string("op_3362")]; tensor var_3364 = mul(x = var_3362_0, y = layers_1_self_attn_k_norm_weight)[name = string("op_3364")]; tensor var_3369 = const()[name = string("op_3369"), val = tensor([1, 1, 1, 256])]; tensor q_53 = reshape(shape = var_3369, x = var_3364)[name = string("q_53")]; fp16 var_3371_promoted = const()[name = string("op_3371_promoted"), val = fp16(0x1p+1)]; tensor var_3340 = transpose(perm = var_3339, x = var_3334)[name = string("transpose_17")]; tensor var_3372 = pow(x = var_3340, y = var_3371_promoted)[name = string("op_3372")]; tensor var_3377_axes_0 = const()[name = string("op_3377_axes_0"), val = tensor([-1])]; bool var_3377_keep_dims_0 = const()[name = string("op_3377_keep_dims_0"), val = bool(true)]; tensor var_3377 = reduce_mean(axes = var_3377_axes_0, keep_dims = var_3377_keep_dims_0, x = var_3372)[name = string("op_3377")]; fp16 var_3379_to_fp16 = const()[name = string("op_3379_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_13_cast_fp16 = add(x = var_3377, y = var_3379_to_fp16)[name = string("mean_sq_13_cast_fp16")]; fp32 var_3381_epsilon_0 = const()[name = string("op_3381_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_3381_cast_fp16 = rsqrt(epsilon = var_3381_epsilon_0, x = mean_sq_13_cast_fp16)[name = string("op_3381_cast_fp16")]; tensor input_191_cast_fp16 = mul(x = var_3340, y = var_3381_cast_fp16)[name = string("input_191_cast_fp16")]; tensor var_3383_cast_fp16 = mul(x = q_53, y = cos_s)[name = string("op_3383_cast_fp16")]; tensor var_3384_split_sizes_0 = const()[name = string("op_3384_split_sizes_0"), val = tensor([128, 128])]; int32 var_3384_axis_0 = const()[name = string("op_3384_axis_0"), val = int32(-1)]; tensor var_3384_0, tensor var_3384_1 = split(axis = var_3384_axis_0, split_sizes = var_3384_split_sizes_0, x = q_53)[name = string("op_3384")]; fp16 const_75_promoted = const()[name = string("const_75_promoted"), val = fp16(-0x1p+0)]; tensor var_3386 = mul(x = var_3384_1, y = const_75_promoted)[name = string("op_3386")]; int32 var_3388 = const()[name = string("op_3388"), val = int32(-1)]; bool var_3389_interleave_0 = const()[name = string("op_3389_interleave_0"), val = bool(false)]; tensor var_3389 = concat(axis = var_3388, interleave = var_3389_interleave_0, values = (var_3386, var_3384_0))[name = string("op_3389")]; tensor var_3390_cast_fp16 = mul(x = var_3389, y = sin_s)[name = string("op_3390_cast_fp16")]; tensor input_189_cast_fp16 = add(x = var_3383_cast_fp16, y = var_3390_cast_fp16)[name = string("input_189_cast_fp16")]; tensor k_padded_11_pad_0 = const()[name = string("k_padded_11_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_11_mode_0 = const()[name = string("k_padded_11_mode_0"), val = string("constant")]; fp16 const_76_to_fp16 = const()[name = string("const_76_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_11_cast_fp16 = pad(constant_val = const_76_to_fp16, mode = k_padded_11_mode_0, pad = k_padded_11_pad_0, x = input_189_cast_fp16)[name = string("k_padded_11_cast_fp16")]; tensor v_padded_11_pad_0 = const()[name = string("v_padded_11_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_11_mode_0 = const()[name = string("v_padded_11_mode_0"), val = string("constant")]; fp16 const_77_to_fp16 = const()[name = string("const_77_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_11_cast_fp16 = pad(constant_val = const_77_to_fp16, mode = v_padded_11_mode_0, pad = v_padded_11_pad_0, x = input_191_cast_fp16)[name = string("v_padded_11_cast_fp16")]; tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([10])]; tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([11])]; int32 concat_74_axis_0 = const()[name = string("concat_74_axis_0"), val = int32(0)]; bool concat_74_interleave_0 = const()[name = string("concat_74_interleave_0"), val = bool(false)]; tensor concat_74 = concat(axis = concat_74_axis_0, interleave = concat_74_interleave_0, values = (expand_dims_72, expand_dims_73, ring_pos, expand_dims_75))[name = string("concat_74")]; tensor concat_75_values1_0 = const()[name = string("concat_75_values1_0"), val = tensor([0])]; tensor concat_75_values3_0 = const()[name = string("concat_75_values3_0"), val = tensor([0])]; int32 concat_75_axis_0 = const()[name = string("concat_75_axis_0"), val = int32(0)]; bool concat_75_interleave_0 = const()[name = string("concat_75_interleave_0"), val = bool(false)]; tensor concat_75 = concat(axis = concat_75_axis_0, interleave = concat_75_interleave_0, values = (expand_dims_76, concat_75_values1_0, var_725, concat_75_values3_0))[name = string("concat_75")]; tensor kv_cache_sliding_internal_tensor_assign_11_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_11_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_11_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_74, begin_mask = kv_cache_sliding_internal_tensor_assign_11_begin_mask_0, end = concat_75, end_mask = kv_cache_sliding_internal_tensor_assign_11_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_11_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_11_stride_0, update = k_padded_11_cast_fp16, x = coreml_update_state_27)[name = string("kv_cache_sliding_internal_tensor_assign_11_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_11_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_12_write_state")]; tensor coreml_update_state_28 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_12")]; tensor expand_dims_78 = const()[name = string("expand_dims_78"), val = tensor([11])]; tensor expand_dims_79 = const()[name = string("expand_dims_79"), val = tensor([0])]; tensor expand_dims_81 = const()[name = string("expand_dims_81"), val = tensor([0])]; tensor expand_dims_82 = const()[name = string("expand_dims_82"), val = tensor([12])]; int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_78, expand_dims_79, ring_pos, expand_dims_81))[name = string("concat_78")]; tensor concat_79_values1_0 = const()[name = string("concat_79_values1_0"), val = tensor([0])]; tensor concat_79_values3_0 = const()[name = string("concat_79_values3_0"), val = tensor([0])]; int32 concat_79_axis_0 = const()[name = string("concat_79_axis_0"), val = int32(0)]; bool concat_79_interleave_0 = const()[name = string("concat_79_interleave_0"), val = bool(false)]; tensor concat_79 = concat(axis = concat_79_axis_0, interleave = concat_79_interleave_0, values = (expand_dims_82, concat_79_values1_0, var_725, concat_79_values3_0))[name = string("concat_79")]; tensor kv_cache_sliding_internal_tensor_assign_12_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_12_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_12_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_78, begin_mask = kv_cache_sliding_internal_tensor_assign_12_begin_mask_0, end = concat_79, end_mask = kv_cache_sliding_internal_tensor_assign_12_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_12_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_12_stride_0, update = v_padded_11_cast_fp16, x = coreml_update_state_28)[name = string("kv_cache_sliding_internal_tensor_assign_12_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_12_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_13_write_state")]; tensor coreml_update_state_29 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_13")]; tensor var_3457_begin_0 = const()[name = string("op_3457_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_3457_end_0 = const()[name = string("op_3457_end_0"), val = tensor([11, 1, 512, 512])]; tensor var_3457_end_mask_0 = const()[name = string("op_3457_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3457_cast_fp16 = slice_by_index(begin = var_3457_begin_0, end = var_3457_end_0, end_mask = var_3457_end_mask_0, x = coreml_update_state_29)[name = string("op_3457_cast_fp16")]; tensor K_sliding_slice_11_begin_0 = const()[name = string("K_sliding_slice_11_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_sliding_slice_11_end_0 = const()[name = string("K_sliding_slice_11_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_sliding_slice_11_end_mask_0 = const()[name = string("K_sliding_slice_11_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_sliding_slice_11_cast_fp16 = slice_by_index(begin = K_sliding_slice_11_begin_0, end = K_sliding_slice_11_end_0, end_mask = K_sliding_slice_11_end_mask_0, x = var_3457_cast_fp16)[name = string("K_sliding_slice_11_cast_fp16")]; tensor var_3477_begin_0 = const()[name = string("op_3477_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_3477_end_0 = const()[name = string("op_3477_end_0"), val = tensor([12, 1, 512, 512])]; tensor var_3477_end_mask_0 = const()[name = string("op_3477_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3477_cast_fp16 = slice_by_index(begin = var_3477_begin_0, end = var_3477_end_0, end_mask = var_3477_end_mask_0, x = coreml_update_state_29)[name = string("op_3477_cast_fp16")]; tensor V_for_attn_11_begin_0 = const()[name = string("V_for_attn_11_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_11_end_0 = const()[name = string("V_for_attn_11_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_11_end_mask_0 = const()[name = string("V_for_attn_11_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_11_cast_fp16 = slice_by_index(begin = V_for_attn_11_begin_0, end = V_for_attn_11_end_0, end_mask = V_for_attn_11_end_mask_0, x = var_3477_cast_fp16)[name = string("V_for_attn_11_cast_fp16")]; tensor transpose_24_perm_0 = const()[name = string("transpose_24_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_12_reps_0 = const()[name = string("tile_12_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_24_cast_fp16 = transpose(perm = transpose_24_perm_0, x = K_sliding_slice_11_cast_fp16)[name = string("transpose_16")]; tensor tile_12_cast_fp16 = tile(reps = tile_12_reps_0, x = transpose_24_cast_fp16)[name = string("tile_12_cast_fp16")]; tensor concat_80 = const()[name = string("concat_80"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_24_cast_fp16 = reshape(shape = concat_80, x = tile_12_cast_fp16)[name = string("reshape_24_cast_fp16")]; tensor transpose_25_perm_0 = const()[name = string("transpose_25_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_81 = const()[name = string("concat_81"), val = tensor([-1, 1, 512, 256])]; tensor transpose_25_cast_fp16 = transpose(perm = transpose_25_perm_0, x = reshape_24_cast_fp16)[name = string("transpose_15")]; tensor reshape_25_cast_fp16 = reshape(shape = concat_81, x = transpose_25_cast_fp16)[name = string("reshape_25_cast_fp16")]; tensor transpose_38_perm_0 = const()[name = string("transpose_38_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_26_perm_0 = const()[name = string("transpose_26_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_13_reps_0 = const()[name = string("tile_13_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_26_cast_fp16 = transpose(perm = transpose_26_perm_0, x = V_for_attn_11_cast_fp16)[name = string("transpose_14")]; tensor tile_13_cast_fp16 = tile(reps = tile_13_reps_0, x = transpose_26_cast_fp16)[name = string("tile_13_cast_fp16")]; tensor concat_82 = const()[name = string("concat_82"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_26_cast_fp16 = reshape(shape = concat_82, x = tile_13_cast_fp16)[name = string("reshape_26_cast_fp16")]; tensor transpose_27_perm_0 = const()[name = string("transpose_27_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_83 = const()[name = string("concat_83"), val = tensor([-1, 1, 512, 256])]; tensor transpose_27_cast_fp16 = transpose(perm = transpose_27_perm_0, x = reshape_26_cast_fp16)[name = string("transpose_13")]; tensor reshape_27_cast_fp16 = reshape(shape = concat_83, x = transpose_27_cast_fp16)[name = string("reshape_27_cast_fp16")]; tensor V_expanded_13_perm_0 = const()[name = string("V_expanded_13_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_25_transpose_x_0 = const()[name = string("attn_weights_25_transpose_x_0"), val = bool(false)]; bool attn_weights_25_transpose_y_0 = const()[name = string("attn_weights_25_transpose_y_0"), val = bool(false)]; tensor transpose_38_cast_fp16 = transpose(perm = transpose_38_perm_0, x = reshape_25_cast_fp16)[name = string("transpose_12")]; tensor attn_weights_25_cast_fp16 = matmul(transpose_x = attn_weights_25_transpose_x_0, transpose_y = attn_weights_25_transpose_y_0, x = q_55_cast_fp16, y = transpose_38_cast_fp16)[name = string("attn_weights_25_cast_fp16")]; tensor x_127_cast_fp16 = add(x = attn_weights_25_cast_fp16, y = causal_mask_sliding)[name = string("x_127_cast_fp16")]; tensor reduce_max_6_axes_0 = const()[name = string("reduce_max_6_axes_0"), val = tensor([-1])]; bool reduce_max_6_keep_dims_0 = const()[name = string("reduce_max_6_keep_dims_0"), val = bool(true)]; tensor reduce_max_6 = reduce_max(axes = reduce_max_6_axes_0, keep_dims = reduce_max_6_keep_dims_0, x = x_127_cast_fp16)[name = string("reduce_max_6")]; tensor var_3522 = sub(x = x_127_cast_fp16, y = reduce_max_6)[name = string("op_3522")]; tensor var_3528 = exp(x = var_3522)[name = string("op_3528")]; tensor var_3538_axes_0 = const()[name = string("op_3538_axes_0"), val = tensor([-1])]; bool var_3538_keep_dims_0 = const()[name = string("op_3538_keep_dims_0"), val = bool(true)]; tensor var_3538 = reduce_sum(axes = var_3538_axes_0, keep_dims = var_3538_keep_dims_0, x = var_3528)[name = string("op_3538")]; tensor var_3544_cast_fp16 = real_div(x = var_3528, y = var_3538)[name = string("op_3544_cast_fp16")]; bool attn_output_25_transpose_x_0 = const()[name = string("attn_output_25_transpose_x_0"), val = bool(false)]; bool attn_output_25_transpose_y_0 = const()[name = string("attn_output_25_transpose_y_0"), val = bool(false)]; tensor V_expanded_13_cast_fp16 = transpose(perm = V_expanded_13_perm_0, x = reshape_27_cast_fp16)[name = string("transpose_11")]; tensor attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_0, transpose_y = attn_output_25_transpose_y_0, x = var_3544_cast_fp16, y = V_expanded_13_cast_fp16)[name = string("attn_output_25_cast_fp16")]; tensor var_3555 = const()[name = string("op_3555"), val = tensor([0, 2, 1, 3])]; tensor var_3562 = const()[name = string("op_3562"), val = tensor([1, 1, -1])]; tensor var_3556_cast_fp16 = transpose(perm = var_3555, x = attn_output_25_cast_fp16)[name = string("transpose_10")]; tensor input_193_cast_fp16 = reshape(shape = var_3562, x = var_3556_cast_fp16)[name = string("input_193_cast_fp16")]; tensor layers_6_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151916864))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153489792))))[name = string("layers_6_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_58_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_6_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_193_cast_fp16)[name = string("linear_58_cast_fp16")]; int32 var_3571 = const()[name = string("op_3571"), val = int32(-1)]; fp16 const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3573_cast_fp16 = mul(x = linear_58_cast_fp16, y = const_78_promoted_to_fp16)[name = string("op_3573_cast_fp16")]; bool input_195_interleave_0 = const()[name = string("input_195_interleave_0"), val = bool(false)]; tensor input_195_cast_fp16 = concat(axis = var_3571, interleave = input_195_interleave_0, values = (linear_58_cast_fp16, var_3573_cast_fp16))[name = string("input_195_cast_fp16")]; tensor normed_185_axes_0 = const()[name = string("normed_185_axes_0"), val = tensor([-1])]; fp16 var_3568_to_fp16 = const()[name = string("op_3568_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_185_cast_fp16 = layer_norm(axes = normed_185_axes_0, epsilon = var_3568_to_fp16, x = input_195_cast_fp16)[name = string("normed_185_cast_fp16")]; tensor var_3578_split_sizes_0 = const()[name = string("op_3578_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3578_axis_0 = const()[name = string("op_3578_axis_0"), val = int32(-1)]; tensor var_3578_cast_fp16_0, tensor var_3578_cast_fp16_1 = split(axis = var_3578_axis_0, split_sizes = var_3578_split_sizes_0, x = normed_185_cast_fp16)[name = string("op_3578_cast_fp16")]; tensor layers_6_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_6_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153491392)))]; tensor attn_output_27_cast_fp16 = mul(x = var_3578_cast_fp16_0, y = layers_6_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_27_cast_fp16")]; tensor x_133_cast_fp16 = add(x = x_119_cast_fp16, y = attn_output_27_cast_fp16)[name = string("x_133_cast_fp16")]; int32 var_3587 = const()[name = string("op_3587"), val = int32(-1)]; fp16 const_79_promoted_to_fp16 = const()[name = string("const_79_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3589_cast_fp16 = mul(x = x_133_cast_fp16, y = const_79_promoted_to_fp16)[name = string("op_3589_cast_fp16")]; bool input_197_interleave_0 = const()[name = string("input_197_interleave_0"), val = bool(false)]; tensor input_197_cast_fp16 = concat(axis = var_3587, interleave = input_197_interleave_0, values = (x_133_cast_fp16, var_3589_cast_fp16))[name = string("input_197_cast_fp16")]; tensor normed_189_axes_0 = const()[name = string("normed_189_axes_0"), val = tensor([-1])]; fp16 var_3584_to_fp16 = const()[name = string("op_3584_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_189_cast_fp16 = layer_norm(axes = normed_189_axes_0, epsilon = var_3584_to_fp16, x = input_197_cast_fp16)[name = string("normed_189_cast_fp16")]; tensor var_3594_split_sizes_0 = const()[name = string("op_3594_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3594_axis_0 = const()[name = string("op_3594_axis_0"), val = int32(-1)]; tensor var_3594_cast_fp16_0, tensor var_3594_cast_fp16_1 = split(axis = var_3594_axis_0, split_sizes = var_3594_split_sizes_0, x = normed_189_cast_fp16)[name = string("op_3594_cast_fp16")]; tensor layers_6_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_6_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153494528)))]; tensor var_3596_cast_fp16 = mul(x = var_3594_cast_fp16_0, y = layers_6_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("op_3596_cast_fp16")]; tensor gate_25 = linear(bias = linear_5_bias_0, weight = layers_6_mlp_gate_proj_weight_palettized, x = var_3596_cast_fp16)[name = string("linear_59")]; tensor up_13 = linear(bias = linear_5_bias_0, weight = layers_6_mlp_up_proj_weight_palettized, x = var_3596_cast_fp16)[name = string("linear_60")]; string gate_27_mode_0 = const()[name = string("gate_27_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_27 = gelu(mode = gate_27_mode_0, x = gate_25)[name = string("gate_27")]; tensor input_201 = mul(x = gate_27, y = up_13)[name = string("input_201")]; tensor x_135 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_6_mlp_down_proj_weight_palettized, x = input_201)[name = string("linear_61")]; int32 var_3618 = const()[name = string("op_3618"), val = int32(-1)]; fp16 const_80_promoted = const()[name = string("const_80_promoted"), val = fp16(-0x1p+0)]; tensor var_3620 = mul(x = x_135, y = const_80_promoted)[name = string("op_3620")]; bool input_203_interleave_0 = const()[name = string("input_203_interleave_0"), val = bool(false)]; tensor input_203 = concat(axis = var_3618, interleave = input_203_interleave_0, values = (x_135, var_3620))[name = string("input_203")]; tensor normed_193_axes_0 = const()[name = string("normed_193_axes_0"), val = tensor([-1])]; fp16 var_3615_to_fp16 = const()[name = string("op_3615_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_193_cast_fp16 = layer_norm(axes = normed_193_axes_0, epsilon = var_3615_to_fp16, x = input_203)[name = string("normed_193_cast_fp16")]; tensor var_3625_split_sizes_0 = const()[name = string("op_3625_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3625_axis_0 = const()[name = string("op_3625_axis_0"), val = int32(-1)]; tensor var_3625_0, tensor var_3625_1 = split(axis = var_3625_axis_0, split_sizes = var_3625_split_sizes_0, x = normed_193_cast_fp16)[name = string("op_3625")]; tensor hidden_states_51 = mul(x = var_3625_0, y = layers_6_post_feedforward_layernorm_weight)[name = string("hidden_states_51")]; tensor hidden_states_53_cast_fp16 = add(x = x_133_cast_fp16, y = hidden_states_51)[name = string("hidden_states_53_cast_fp16")]; tensor per_layer_slice_13_begin_0 = const()[name = string("per_layer_slice_13_begin_0"), val = tensor([0, 0, 1536])]; tensor per_layer_slice_13_end_0 = const()[name = string("per_layer_slice_13_end_0"), val = tensor([1, 1, 1792])]; tensor per_layer_slice_13_end_mask_0 = const()[name = string("per_layer_slice_13_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_13_cast_fp16 = slice_by_index(begin = per_layer_slice_13_begin_0, end = per_layer_slice_13_end_0, end_mask = per_layer_slice_13_end_mask_0, x = per_layer_combined_out)[name = string("per_layer_slice_13_cast_fp16")]; tensor gated_25 = linear(bias = linear_2_bias_0, weight = layers_6_per_layer_input_gate_weight_palettized, x = hidden_states_53_cast_fp16)[name = string("linear_62")]; string gated_27_mode_0 = const()[name = string("gated_27_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_27 = gelu(mode = gated_27_mode_0, x = gated_25)[name = string("gated_27")]; tensor input_207_cast_fp16 = mul(x = gated_27, y = per_layer_slice_13_cast_fp16)[name = string("input_207_cast_fp16")]; tensor layers_6_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153497664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153694336))))[name = string("layers_6_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_63_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_6_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_207_cast_fp16)[name = string("linear_63_cast_fp16")]; int32 var_3663 = const()[name = string("op_3663"), val = int32(-1)]; fp16 const_81_promoted_to_fp16 = const()[name = string("const_81_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3665_cast_fp16 = mul(x = linear_63_cast_fp16, y = const_81_promoted_to_fp16)[name = string("op_3665_cast_fp16")]; bool input_209_interleave_0 = const()[name = string("input_209_interleave_0"), val = bool(false)]; tensor input_209_cast_fp16 = concat(axis = var_3663, interleave = input_209_interleave_0, values = (linear_63_cast_fp16, var_3665_cast_fp16))[name = string("input_209_cast_fp16")]; tensor normed_197_axes_0 = const()[name = string("normed_197_axes_0"), val = tensor([-1])]; fp16 var_3660_to_fp16 = const()[name = string("op_3660_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_197_cast_fp16 = layer_norm(axes = normed_197_axes_0, epsilon = var_3660_to_fp16, x = input_209_cast_fp16)[name = string("normed_197_cast_fp16")]; tensor var_3670_split_sizes_0 = const()[name = string("op_3670_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3670_axis_0 = const()[name = string("op_3670_axis_0"), val = int32(-1)]; tensor var_3670_cast_fp16_0, tensor var_3670_cast_fp16_1 = split(axis = var_3670_axis_0, split_sizes = var_3670_split_sizes_0, x = normed_197_cast_fp16)[name = string("op_3670_cast_fp16")]; tensor layers_6_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_6_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153695936)))]; tensor hidden_states_55_cast_fp16 = mul(x = var_3670_cast_fp16_0, y = layers_6_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_55_cast_fp16")]; tensor hidden_states_57_cast_fp16 = add(x = hidden_states_53_cast_fp16, y = hidden_states_55_cast_fp16)[name = string("hidden_states_57_cast_fp16")]; tensor const_82_promoted_to_fp16 = const()[name = string("const_82_promoted_to_fp16"), val = tensor([0x1.fep-2])]; tensor x_139_cast_fp16 = mul(x = hidden_states_57_cast_fp16, y = const_82_promoted_to_fp16)[name = string("x_139_cast_fp16")]; int32 var_3685 = const()[name = string("op_3685"), val = int32(-1)]; fp16 const_83_promoted_to_fp16 = const()[name = string("const_83_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3687_cast_fp16 = mul(x = x_139_cast_fp16, y = const_83_promoted_to_fp16)[name = string("op_3687_cast_fp16")]; bool input_211_interleave_0 = const()[name = string("input_211_interleave_0"), val = bool(false)]; tensor input_211_cast_fp16 = concat(axis = var_3685, interleave = input_211_interleave_0, values = (x_139_cast_fp16, var_3687_cast_fp16))[name = string("input_211_cast_fp16")]; tensor normed_201_axes_0 = const()[name = string("normed_201_axes_0"), val = tensor([-1])]; fp16 var_3682_to_fp16 = const()[name = string("op_3682_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_201_cast_fp16 = layer_norm(axes = normed_201_axes_0, epsilon = var_3682_to_fp16, x = input_211_cast_fp16)[name = string("normed_201_cast_fp16")]; tensor var_3692_split_sizes_0 = const()[name = string("op_3692_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3692_axis_0 = const()[name = string("op_3692_axis_0"), val = int32(-1)]; tensor var_3692_cast_fp16_0, tensor var_3692_cast_fp16_1 = split(axis = var_3692_axis_0, split_sizes = var_3692_split_sizes_0, x = normed_201_cast_fp16)[name = string("op_3692_cast_fp16")]; tensor layers_7_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_7_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153699072)))]; tensor var_3694_cast_fp16 = mul(x = var_3692_cast_fp16_0, y = layers_7_input_layernorm_weight_promoted_to_fp16)[name = string("op_3694_cast_fp16")]; tensor var_3702 = linear(bias = linear_1_bias_0, weight = layers_7_self_attn_q_proj_weight_palettized, x = var_3694_cast_fp16)[name = string("linear_64")]; tensor var_3707 = const()[name = string("op_3707"), val = tensor([1, 8, 256, 1])]; tensor var_3708 = reshape(shape = var_3707, x = var_3702)[name = string("op_3708")]; tensor var_3713 = const()[name = string("op_3713"), val = tensor([0, 1, 3, 2])]; tensor var_3723 = const()[name = string("op_3723"), val = tensor([1, 8, 256])]; tensor var_3714 = transpose(perm = var_3713, x = var_3708)[name = string("transpose_9")]; tensor x_141 = reshape(shape = var_3723, x = var_3714)[name = string("x_141")]; int32 var_3729 = const()[name = string("op_3729"), val = int32(-1)]; fp16 const_84_promoted = const()[name = string("const_84_promoted"), val = fp16(-0x1p+0)]; tensor var_3731 = mul(x = x_141, y = const_84_promoted)[name = string("op_3731")]; bool input_215_interleave_0 = const()[name = string("input_215_interleave_0"), val = bool(false)]; tensor input_215 = concat(axis = var_3729, interleave = input_215_interleave_0, values = (x_141, var_3731))[name = string("input_215")]; tensor normed_205_axes_0 = const()[name = string("normed_205_axes_0"), val = tensor([-1])]; fp16 var_3726_to_fp16 = const()[name = string("op_3726_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_205_cast_fp16 = layer_norm(axes = normed_205_axes_0, epsilon = var_3726_to_fp16, x = input_215)[name = string("normed_205_cast_fp16")]; tensor var_3736_split_sizes_0 = const()[name = string("op_3736_split_sizes_0"), val = tensor([256, 256])]; int32 var_3736_axis_0 = const()[name = string("op_3736_axis_0"), val = int32(-1)]; tensor var_3736_0, tensor var_3736_1 = split(axis = var_3736_axis_0, split_sizes = var_3736_split_sizes_0, x = normed_205_cast_fp16)[name = string("op_3736")]; tensor var_3738 = mul(x = var_3736_0, y = layers_7_self_attn_q_norm_weight)[name = string("op_3738")]; tensor var_3743 = const()[name = string("op_3743"), val = tensor([1, 8, 1, 256])]; tensor q_59 = reshape(shape = var_3743, x = var_3738)[name = string("q_59")]; tensor var_3745_cast_fp16 = mul(x = q_59, y = cos_s)[name = string("op_3745_cast_fp16")]; tensor var_3746_split_sizes_0 = const()[name = string("op_3746_split_sizes_0"), val = tensor([128, 128])]; int32 var_3746_axis_0 = const()[name = string("op_3746_axis_0"), val = int32(-1)]; tensor var_3746_0, tensor var_3746_1 = split(axis = var_3746_axis_0, split_sizes = var_3746_split_sizes_0, x = q_59)[name = string("op_3746")]; fp16 const_85_promoted = const()[name = string("const_85_promoted"), val = fp16(-0x1p+0)]; tensor var_3748 = mul(x = var_3746_1, y = const_85_promoted)[name = string("op_3748")]; int32 var_3750 = const()[name = string("op_3750"), val = int32(-1)]; bool var_3751_interleave_0 = const()[name = string("op_3751_interleave_0"), val = bool(false)]; tensor var_3751 = concat(axis = var_3750, interleave = var_3751_interleave_0, values = (var_3748, var_3746_0))[name = string("op_3751")]; tensor var_3752_cast_fp16 = mul(x = var_3751, y = sin_s)[name = string("op_3752_cast_fp16")]; tensor q_cast_fp16 = add(x = var_3745_cast_fp16, y = var_3752_cast_fp16)[name = string("q_cast_fp16")]; tensor var_3757 = linear(bias = linear_2_bias_0, weight = layers_7_self_attn_k_proj_weight_palettized, x = var_3694_cast_fp16)[name = string("linear_65")]; tensor var_3762 = const()[name = string("op_3762"), val = tensor([1, 1, 256, 1])]; tensor var_3763 = reshape(shape = var_3762, x = var_3757)[name = string("op_3763")]; tensor var_3768 = const()[name = string("op_3768"), val = tensor([0, 1, 3, 2])]; tensor var_3777 = linear(bias = linear_2_bias_0, weight = layers_7_self_attn_v_proj_weight_palettized, x = var_3694_cast_fp16)[name = string("linear_66")]; tensor var_3782 = const()[name = string("op_3782"), val = tensor([1, 1, 256, 1])]; tensor var_3783 = reshape(shape = var_3782, x = var_3777)[name = string("op_3783")]; tensor var_3788 = const()[name = string("op_3788"), val = tensor([0, 1, 3, 2])]; tensor var_3798 = const()[name = string("op_3798"), val = tensor([1, 1, 256])]; tensor var_3769 = transpose(perm = var_3768, x = var_3763)[name = string("transpose_8")]; tensor x_143 = reshape(shape = var_3798, x = var_3769)[name = string("x_143")]; int32 var_3804 = const()[name = string("op_3804"), val = int32(-1)]; fp16 const_86_promoted = const()[name = string("const_86_promoted"), val = fp16(-0x1p+0)]; tensor var_3806 = mul(x = x_143, y = const_86_promoted)[name = string("op_3806")]; bool input_217_interleave_0 = const()[name = string("input_217_interleave_0"), val = bool(false)]; tensor input_217 = concat(axis = var_3804, interleave = input_217_interleave_0, values = (x_143, var_3806))[name = string("input_217")]; tensor normed_209_axes_0 = const()[name = string("normed_209_axes_0"), val = tensor([-1])]; fp16 var_3801_to_fp16 = const()[name = string("op_3801_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_209_cast_fp16 = layer_norm(axes = normed_209_axes_0, epsilon = var_3801_to_fp16, x = input_217)[name = string("normed_209_cast_fp16")]; tensor var_3811_split_sizes_0 = const()[name = string("op_3811_split_sizes_0"), val = tensor([256, 256])]; int32 var_3811_axis_0 = const()[name = string("op_3811_axis_0"), val = int32(-1)]; tensor var_3811_0, tensor var_3811_1 = split(axis = var_3811_axis_0, split_sizes = var_3811_split_sizes_0, x = normed_209_cast_fp16)[name = string("op_3811")]; tensor var_3813 = mul(x = var_3811_0, y = layers_7_self_attn_k_norm_weight)[name = string("op_3813")]; tensor var_3818 = const()[name = string("op_3818"), val = tensor([1, 1, 1, 256])]; tensor q_61 = reshape(shape = var_3818, x = var_3813)[name = string("q_61")]; fp16 var_3820_promoted = const()[name = string("op_3820_promoted"), val = fp16(0x1p+1)]; tensor var_3789 = transpose(perm = var_3788, x = var_3783)[name = string("transpose_7")]; tensor var_3821 = pow(x = var_3789, y = var_3820_promoted)[name = string("op_3821")]; tensor var_3826_axes_0 = const()[name = string("op_3826_axes_0"), val = tensor([-1])]; bool var_3826_keep_dims_0 = const()[name = string("op_3826_keep_dims_0"), val = bool(true)]; tensor var_3826 = reduce_mean(axes = var_3826_axes_0, keep_dims = var_3826_keep_dims_0, x = var_3821)[name = string("op_3826")]; fp16 var_3828_to_fp16 = const()[name = string("op_3828_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_cast_fp16 = add(x = var_3826, y = var_3828_to_fp16)[name = string("mean_sq_cast_fp16")]; fp32 var_3830_epsilon_0 = const()[name = string("op_3830_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_3830_cast_fp16 = rsqrt(epsilon = var_3830_epsilon_0, x = mean_sq_cast_fp16)[name = string("op_3830_cast_fp16")]; tensor input_221_cast_fp16 = mul(x = var_3789, y = var_3830_cast_fp16)[name = string("input_221_cast_fp16")]; tensor var_3832_cast_fp16 = mul(x = q_61, y = cos_s)[name = string("op_3832_cast_fp16")]; tensor var_3833_split_sizes_0 = const()[name = string("op_3833_split_sizes_0"), val = tensor([128, 128])]; int32 var_3833_axis_0 = const()[name = string("op_3833_axis_0"), val = int32(-1)]; tensor var_3833_0, tensor var_3833_1 = split(axis = var_3833_axis_0, split_sizes = var_3833_split_sizes_0, x = q_61)[name = string("op_3833")]; fp16 const_87_promoted = const()[name = string("const_87_promoted"), val = fp16(-0x1p+0)]; tensor var_3835 = mul(x = var_3833_1, y = const_87_promoted)[name = string("op_3835")]; int32 var_3837 = const()[name = string("op_3837"), val = int32(-1)]; bool var_3838_interleave_0 = const()[name = string("op_3838_interleave_0"), val = bool(false)]; tensor var_3838 = concat(axis = var_3837, interleave = var_3838_interleave_0, values = (var_3835, var_3833_0))[name = string("op_3838")]; tensor var_3839_cast_fp16 = mul(x = var_3838, y = sin_s)[name = string("op_3839_cast_fp16")]; tensor input_219_cast_fp16 = add(x = var_3832_cast_fp16, y = var_3839_cast_fp16)[name = string("input_219_cast_fp16")]; tensor k_padded_pad_0 = const()[name = string("k_padded_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_mode_0 = const()[name = string("k_padded_mode_0"), val = string("constant")]; fp16 const_88_to_fp16 = const()[name = string("const_88_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_cast_fp16 = pad(constant_val = const_88_to_fp16, mode = k_padded_mode_0, pad = k_padded_pad_0, x = input_219_cast_fp16)[name = string("k_padded_cast_fp16")]; tensor v_padded_pad_0 = const()[name = string("v_padded_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_mode_0 = const()[name = string("v_padded_mode_0"), val = string("constant")]; fp16 const_89_to_fp16 = const()[name = string("const_89_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_cast_fp16 = pad(constant_val = const_89_to_fp16, mode = v_padded_mode_0, pad = v_padded_pad_0, x = input_221_cast_fp16)[name = string("v_padded_cast_fp16")]; tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([12])]; tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([13])]; int32 concat_86_axis_0 = const()[name = string("concat_86_axis_0"), val = int32(0)]; bool concat_86_interleave_0 = const()[name = string("concat_86_interleave_0"), val = bool(false)]; tensor concat_86 = concat(axis = concat_86_axis_0, interleave = concat_86_interleave_0, values = (expand_dims_84, expand_dims_85, ring_pos, expand_dims_87))[name = string("concat_86")]; tensor concat_87_values1_0 = const()[name = string("concat_87_values1_0"), val = tensor([0])]; tensor concat_87_values3_0 = const()[name = string("concat_87_values3_0"), val = tensor([0])]; int32 concat_87_axis_0 = const()[name = string("concat_87_axis_0"), val = int32(0)]; bool concat_87_interleave_0 = const()[name = string("concat_87_interleave_0"), val = bool(false)]; tensor concat_87 = concat(axis = concat_87_axis_0, interleave = concat_87_interleave_0, values = (expand_dims_88, concat_87_values1_0, var_725, concat_87_values3_0))[name = string("concat_87")]; tensor kv_cache_sliding_internal_tensor_assign_13_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_13_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_13_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_86, begin_mask = kv_cache_sliding_internal_tensor_assign_13_begin_mask_0, end = concat_87, end_mask = kv_cache_sliding_internal_tensor_assign_13_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_13_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_13_stride_0, update = k_padded_cast_fp16, x = coreml_update_state_29)[name = string("kv_cache_sliding_internal_tensor_assign_13_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_13_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_14_write_state")]; tensor coreml_update_state_30 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_14")]; tensor expand_dims_90 = const()[name = string("expand_dims_90"), val = tensor([13])]; tensor expand_dims_91 = const()[name = string("expand_dims_91"), val = tensor([0])]; tensor expand_dims_93 = const()[name = string("expand_dims_93"), val = tensor([0])]; tensor expand_dims_94 = const()[name = string("expand_dims_94"), val = tensor([14])]; int32 concat_90_axis_0 = const()[name = string("concat_90_axis_0"), val = int32(0)]; bool concat_90_interleave_0 = const()[name = string("concat_90_interleave_0"), val = bool(false)]; tensor concat_90 = concat(axis = concat_90_axis_0, interleave = concat_90_interleave_0, values = (expand_dims_90, expand_dims_91, ring_pos, expand_dims_93))[name = string("concat_90")]; tensor concat_91_values1_0 = const()[name = string("concat_91_values1_0"), val = tensor([0])]; tensor concat_91_values3_0 = const()[name = string("concat_91_values3_0"), val = tensor([0])]; int32 concat_91_axis_0 = const()[name = string("concat_91_axis_0"), val = int32(0)]; bool concat_91_interleave_0 = const()[name = string("concat_91_interleave_0"), val = bool(false)]; tensor concat_91 = concat(axis = concat_91_axis_0, interleave = concat_91_interleave_0, values = (expand_dims_94, concat_91_values1_0, var_725, concat_91_values3_0))[name = string("concat_91")]; tensor kv_cache_sliding_internal_tensor_assign_14_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_14_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_14_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_90, begin_mask = kv_cache_sliding_internal_tensor_assign_14_begin_mask_0, end = concat_91, end_mask = kv_cache_sliding_internal_tensor_assign_14_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_14_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_14_stride_0, update = v_padded_cast_fp16, x = coreml_update_state_30)[name = string("kv_cache_sliding_internal_tensor_assign_14_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_14_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_15_write_state")]; tensor coreml_update_state_31 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_15")]; tensor var_3906_begin_0 = const()[name = string("op_3906_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_3906_end_0 = const()[name = string("op_3906_end_0"), val = tensor([13, 1, 512, 512])]; tensor var_3906_end_mask_0 = const()[name = string("op_3906_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3906_cast_fp16 = slice_by_index(begin = var_3906_begin_0, end = var_3906_end_0, end_mask = var_3906_end_mask_0, x = coreml_update_state_31)[name = string("op_3906_cast_fp16")]; tensor K_sliding_slice_begin_0 = const()[name = string("K_sliding_slice_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_sliding_slice_end_0 = const()[name = string("K_sliding_slice_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_sliding_slice_end_mask_0 = const()[name = string("K_sliding_slice_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_sliding_slice_cast_fp16 = slice_by_index(begin = K_sliding_slice_begin_0, end = K_sliding_slice_end_0, end_mask = K_sliding_slice_end_mask_0, x = var_3906_cast_fp16)[name = string("K_sliding_slice_cast_fp16")]; tensor var_3926_begin_0 = const()[name = string("op_3926_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_3926_end_0 = const()[name = string("op_3926_end_0"), val = tensor([1, 1, 512, 512])]; tensor var_3926_end_mask_0 = const()[name = string("op_3926_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3926_cast_fp16 = slice_by_index(begin = var_3926_begin_0, end = var_3926_end_0, end_mask = var_3926_end_mask_0, x = coreml_update_state_31)[name = string("op_3926_cast_fp16")]; tensor V_for_attn_begin_0 = const()[name = string("V_for_attn_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_end_0 = const()[name = string("V_for_attn_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_end_mask_0 = const()[name = string("V_for_attn_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_cast_fp16 = slice_by_index(begin = V_for_attn_begin_0, end = V_for_attn_end_0, end_mask = V_for_attn_end_mask_0, x = var_3926_cast_fp16)[name = string("V_for_attn_cast_fp16")]; tensor transpose_28_perm_0 = const()[name = string("transpose_28_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_14_reps_0 = const()[name = string("tile_14_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_28_cast_fp16 = transpose(perm = transpose_28_perm_0, x = K_sliding_slice_cast_fp16)[name = string("transpose_6")]; tensor tile_14_cast_fp16 = tile(reps = tile_14_reps_0, x = transpose_28_cast_fp16)[name = string("tile_14_cast_fp16")]; tensor concat_92 = const()[name = string("concat_92"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_28_cast_fp16 = reshape(shape = concat_92, x = tile_14_cast_fp16)[name = string("reshape_28_cast_fp16")]; tensor transpose_29_perm_0 = const()[name = string("transpose_29_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_93 = const()[name = string("concat_93"), val = tensor([-1, 1, 512, 256])]; tensor transpose_29_cast_fp16 = transpose(perm = transpose_29_perm_0, x = reshape_28_cast_fp16)[name = string("transpose_5")]; tensor reshape_29_cast_fp16 = reshape(shape = concat_93, x = transpose_29_cast_fp16)[name = string("reshape_29_cast_fp16")]; tensor transpose_39_perm_0 = const()[name = string("transpose_39_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_30_perm_0 = const()[name = string("transpose_30_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_15_reps_0 = const()[name = string("tile_15_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_30_cast_fp16 = transpose(perm = transpose_30_perm_0, x = V_for_attn_cast_fp16)[name = string("transpose_4")]; tensor tile_15_cast_fp16 = tile(reps = tile_15_reps_0, x = transpose_30_cast_fp16)[name = string("tile_15_cast_fp16")]; tensor concat_94 = const()[name = string("concat_94"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_30_cast_fp16 = reshape(shape = concat_94, x = tile_15_cast_fp16)[name = string("reshape_30_cast_fp16")]; tensor transpose_31_perm_0 = const()[name = string("transpose_31_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_95 = const()[name = string("concat_95"), val = tensor([-1, 1, 512, 256])]; tensor transpose_31_cast_fp16 = transpose(perm = transpose_31_perm_0, x = reshape_30_cast_fp16)[name = string("transpose_3")]; tensor reshape_31_cast_fp16 = reshape(shape = concat_95, x = transpose_31_cast_fp16)[name = string("reshape_31_cast_fp16")]; tensor V_expanded_perm_0 = const()[name = string("V_expanded_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_29_transpose_x_0 = const()[name = string("attn_weights_29_transpose_x_0"), val = bool(false)]; bool attn_weights_29_transpose_y_0 = const()[name = string("attn_weights_29_transpose_y_0"), val = bool(false)]; tensor transpose_39_cast_fp16 = transpose(perm = transpose_39_perm_0, x = reshape_29_cast_fp16)[name = string("transpose_2")]; tensor attn_weights_29_cast_fp16 = matmul(transpose_x = attn_weights_29_transpose_x_0, transpose_y = attn_weights_29_transpose_y_0, x = q_cast_fp16, y = transpose_39_cast_fp16)[name = string("attn_weights_29_cast_fp16")]; tensor x_147_cast_fp16 = add(x = attn_weights_29_cast_fp16, y = causal_mask_sliding)[name = string("x_147_cast_fp16")]; tensor reduce_max_7_axes_0 = const()[name = string("reduce_max_7_axes_0"), val = tensor([-1])]; bool reduce_max_7_keep_dims_0 = const()[name = string("reduce_max_7_keep_dims_0"), val = bool(true)]; tensor reduce_max_7 = reduce_max(axes = reduce_max_7_axes_0, keep_dims = reduce_max_7_keep_dims_0, x = x_147_cast_fp16)[name = string("reduce_max_7")]; tensor var_3971 = sub(x = x_147_cast_fp16, y = reduce_max_7)[name = string("op_3971")]; tensor var_3977 = exp(x = var_3971)[name = string("op_3977")]; tensor var_3987_axes_0 = const()[name = string("op_3987_axes_0"), val = tensor([-1])]; bool var_3987_keep_dims_0 = const()[name = string("op_3987_keep_dims_0"), val = bool(true)]; tensor var_3987 = reduce_sum(axes = var_3987_axes_0, keep_dims = var_3987_keep_dims_0, x = var_3977)[name = string("op_3987")]; tensor var_3993_cast_fp16 = real_div(x = var_3977, y = var_3987)[name = string("op_3993_cast_fp16")]; bool attn_output_29_transpose_x_0 = const()[name = string("attn_output_29_transpose_x_0"), val = bool(false)]; bool attn_output_29_transpose_y_0 = const()[name = string("attn_output_29_transpose_y_0"), val = bool(false)]; tensor V_expanded_cast_fp16 = transpose(perm = V_expanded_perm_0, x = reshape_31_cast_fp16)[name = string("transpose_1")]; tensor attn_output_29_cast_fp16 = matmul(transpose_x = attn_output_29_transpose_x_0, transpose_y = attn_output_29_transpose_y_0, x = var_3993_cast_fp16, y = V_expanded_cast_fp16)[name = string("attn_output_29_cast_fp16")]; tensor var_4004 = const()[name = string("op_4004"), val = tensor([0, 2, 1, 3])]; tensor var_4011 = const()[name = string("op_4011"), val = tensor([1, 1, -1])]; tensor var_4005_cast_fp16 = transpose(perm = var_4004, x = attn_output_29_cast_fp16)[name = string("transpose_0")]; tensor input_223_cast_fp16 = reshape(shape = var_4011, x = var_4005_cast_fp16)[name = string("input_223_cast_fp16")]; tensor layers_7_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153702208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(155275136))))[name = string("layers_7_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_67_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_7_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_223_cast_fp16)[name = string("linear_67_cast_fp16")]; int32 var_4020 = const()[name = string("op_4020"), val = int32(-1)]; fp16 const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4022_cast_fp16 = mul(x = linear_67_cast_fp16, y = const_90_promoted_to_fp16)[name = string("op_4022_cast_fp16")]; bool input_225_interleave_0 = const()[name = string("input_225_interleave_0"), val = bool(false)]; tensor input_225_cast_fp16 = concat(axis = var_4020, interleave = input_225_interleave_0, values = (linear_67_cast_fp16, var_4022_cast_fp16))[name = string("input_225_cast_fp16")]; tensor normed_213_axes_0 = const()[name = string("normed_213_axes_0"), val = tensor([-1])]; fp16 var_4017_to_fp16 = const()[name = string("op_4017_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_213_cast_fp16 = layer_norm(axes = normed_213_axes_0, epsilon = var_4017_to_fp16, x = input_225_cast_fp16)[name = string("normed_213_cast_fp16")]; tensor var_4027_split_sizes_0 = const()[name = string("op_4027_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4027_axis_0 = const()[name = string("op_4027_axis_0"), val = int32(-1)]; tensor var_4027_cast_fp16_0, tensor var_4027_cast_fp16_1 = split(axis = var_4027_axis_0, split_sizes = var_4027_split_sizes_0, x = normed_213_cast_fp16)[name = string("op_4027_cast_fp16")]; tensor layers_7_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_7_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(155276736)))]; tensor attn_output_cast_fp16 = mul(x = var_4027_cast_fp16_0, y = layers_7_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_cast_fp16")]; tensor x_153_cast_fp16 = add(x = x_139_cast_fp16, y = attn_output_cast_fp16)[name = string("x_153_cast_fp16")]; int32 var_4036 = const()[name = string("op_4036"), val = int32(-1)]; fp16 const_91_promoted_to_fp16 = const()[name = string("const_91_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4038_cast_fp16 = mul(x = x_153_cast_fp16, y = const_91_promoted_to_fp16)[name = string("op_4038_cast_fp16")]; bool input_227_interleave_0 = const()[name = string("input_227_interleave_0"), val = bool(false)]; tensor input_227_cast_fp16 = concat(axis = var_4036, interleave = input_227_interleave_0, values = (x_153_cast_fp16, var_4038_cast_fp16))[name = string("input_227_cast_fp16")]; tensor normed_217_axes_0 = const()[name = string("normed_217_axes_0"), val = tensor([-1])]; fp16 var_4033_to_fp16 = const()[name = string("op_4033_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_217_cast_fp16 = layer_norm(axes = normed_217_axes_0, epsilon = var_4033_to_fp16, x = input_227_cast_fp16)[name = string("normed_217_cast_fp16")]; tensor var_4043_split_sizes_0 = const()[name = string("op_4043_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4043_axis_0 = const()[name = string("op_4043_axis_0"), val = int32(-1)]; tensor var_4043_cast_fp16_0, tensor var_4043_cast_fp16_1 = split(axis = var_4043_axis_0, split_sizes = var_4043_split_sizes_0, x = normed_217_cast_fp16)[name = string("op_4043_cast_fp16")]; tensor layers_7_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_7_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(155279872)))]; tensor var_4045_cast_fp16 = mul(x = var_4043_cast_fp16_0, y = layers_7_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("op_4045_cast_fp16")]; tensor gate_29 = linear(bias = linear_5_bias_0, weight = layers_7_mlp_gate_proj_weight_palettized, x = var_4045_cast_fp16)[name = string("linear_68")]; tensor up = linear(bias = linear_5_bias_0, weight = layers_7_mlp_up_proj_weight_palettized, x = var_4045_cast_fp16)[name = string("linear_69")]; string gate_mode_0 = const()[name = string("gate_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate = gelu(mode = gate_mode_0, x = gate_29)[name = string("gate")]; tensor input_231 = mul(x = gate, y = up)[name = string("input_231")]; tensor x_155 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_7_mlp_down_proj_weight_palettized, x = input_231)[name = string("linear_70")]; int32 var_4067 = const()[name = string("op_4067"), val = int32(-1)]; fp16 const_92_promoted = const()[name = string("const_92_promoted"), val = fp16(-0x1p+0)]; tensor var_4069 = mul(x = x_155, y = const_92_promoted)[name = string("op_4069")]; bool input_233_interleave_0 = const()[name = string("input_233_interleave_0"), val = bool(false)]; tensor input_233 = concat(axis = var_4067, interleave = input_233_interleave_0, values = (x_155, var_4069))[name = string("input_233")]; tensor normed_221_axes_0 = const()[name = string("normed_221_axes_0"), val = tensor([-1])]; fp16 var_4064_to_fp16 = const()[name = string("op_4064_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_221_cast_fp16 = layer_norm(axes = normed_221_axes_0, epsilon = var_4064_to_fp16, x = input_233)[name = string("normed_221_cast_fp16")]; tensor var_4074_split_sizes_0 = const()[name = string("op_4074_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4074_axis_0 = const()[name = string("op_4074_axis_0"), val = int32(-1)]; tensor var_4074_0, tensor var_4074_1 = split(axis = var_4074_axis_0, split_sizes = var_4074_split_sizes_0, x = normed_221_cast_fp16)[name = string("op_4074")]; tensor hidden_states_59 = mul(x = var_4074_0, y = layers_7_post_feedforward_layernorm_weight)[name = string("hidden_states_59")]; tensor hidden_states_61_cast_fp16 = add(x = x_153_cast_fp16, y = hidden_states_59)[name = string("hidden_states_61_cast_fp16")]; tensor per_layer_slice_begin_0 = const()[name = string("per_layer_slice_begin_0"), val = tensor([0, 0, 1792])]; tensor per_layer_slice_end_0 = const()[name = string("per_layer_slice_end_0"), val = tensor([1, 1, 2048])]; tensor per_layer_slice_end_mask_0 = const()[name = string("per_layer_slice_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_cast_fp16 = slice_by_index(begin = per_layer_slice_begin_0, end = per_layer_slice_end_0, end_mask = per_layer_slice_end_mask_0, x = per_layer_combined_out)[name = string("per_layer_slice_cast_fp16")]; tensor gated_29 = linear(bias = linear_2_bias_0, weight = layers_7_per_layer_input_gate_weight_palettized, x = hidden_states_61_cast_fp16)[name = string("linear_71")]; string gated_mode_0 = const()[name = string("gated_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated = gelu(mode = gated_mode_0, x = gated_29)[name = string("gated")]; tensor input_237_cast_fp16 = mul(x = gated, y = per_layer_slice_cast_fp16)[name = string("input_237_cast_fp16")]; tensor layers_7_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(155283008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(155479680))))[name = string("layers_7_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_72_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_7_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_237_cast_fp16)[name = string("linear_72_cast_fp16")]; int32 var_4112 = const()[name = string("op_4112"), val = int32(-1)]; fp16 const_93_promoted_to_fp16 = const()[name = string("const_93_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4114_cast_fp16 = mul(x = linear_72_cast_fp16, y = const_93_promoted_to_fp16)[name = string("op_4114_cast_fp16")]; bool input_interleave_0 = const()[name = string("input_interleave_0"), val = bool(false)]; tensor input_cast_fp16 = concat(axis = var_4112, interleave = input_interleave_0, values = (linear_72_cast_fp16, var_4114_cast_fp16))[name = string("input_cast_fp16")]; tensor normed_225_axes_0 = const()[name = string("normed_225_axes_0"), val = tensor([-1])]; fp16 var_4109_to_fp16 = const()[name = string("op_4109_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_225_cast_fp16 = layer_norm(axes = normed_225_axes_0, epsilon = var_4109_to_fp16, x = input_cast_fp16)[name = string("normed_225_cast_fp16")]; tensor var_4119_split_sizes_0 = const()[name = string("op_4119_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4119_axis_0 = const()[name = string("op_4119_axis_0"), val = int32(-1)]; tensor var_4119_cast_fp16_0, tensor var_4119_cast_fp16_1 = split(axis = var_4119_axis_0, split_sizes = var_4119_split_sizes_0, x = normed_225_cast_fp16)[name = string("op_4119_cast_fp16")]; tensor layers_7_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_7_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(155481280)))]; tensor hidden_states_63_cast_fp16 = mul(x = var_4119_cast_fp16_0, y = layers_7_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_63_cast_fp16")]; tensor hidden_states_cast_fp16 = add(x = hidden_states_61_cast_fp16, y = hidden_states_63_cast_fp16)[name = string("hidden_states_cast_fp16")]; tensor const_94_promoted_to_fp16 = const()[name = string("const_94_promoted_to_fp16"), val = tensor([0x1.38p-1])]; tensor hidden_states_out = mul(x = hidden_states_cast_fp16, y = const_94_promoted_to_fp16)[name = string("op_4129_cast_fp16")]; } -> (hidden_states_out, per_layer_combined_out); func prefill_b8(tensor causal_mask_full, tensor causal_mask_sliding, tensor cos_f, tensor cos_s, tensor current_pos, tensor hidden_states, state> kv_cache_full, state> kv_cache_sliding, tensor per_layer_raw, tensor ring_pos, tensor sin_f, tensor sin_s) { tensor per_layer_model_projection_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6881408))))[name = string("per_layer_model_projection_weight_palettized")]; tensor layers_0_input_layernorm_weight = const()[name = string("layers_0_input_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6890432)))]; tensor layers_0_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(6893568))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8466496))))[name = string("layers_0_self_attn_q_proj_weight_palettized")]; tensor layers_0_self_attn_q_norm_weight = const()[name = string("layers_0_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8468608)))]; tensor layers_0_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8469184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8665856))))[name = string("layers_0_self_attn_k_proj_weight_palettized")]; tensor layers_0_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8666176))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8862848))))[name = string("layers_0_self_attn_v_proj_weight_palettized")]; tensor layers_0_self_attn_k_norm_weight = const()[name = string("layers_0_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8863168)))]; tensor layers_0_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(8863744))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13582400))))[name = string("layers_0_mlp_gate_proj_weight_palettized")]; tensor layers_0_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13588608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18307264))))[name = string("layers_0_mlp_up_proj_weight_palettized")]; tensor layers_0_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(18313472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23032128))))[name = string("layers_0_mlp_down_proj_weight_palettized")]; tensor layers_0_post_feedforward_layernorm_weight = const()[name = string("layers_0_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23033728)))]; tensor layers_0_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23036864))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23233536))))[name = string("layers_0_per_layer_input_gate_weight_palettized")]; tensor layers_1_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(23233856))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24806784))))[name = string("layers_1_self_attn_q_proj_weight_palettized")]; tensor layers_1_self_attn_q_norm_weight = const()[name = string("layers_1_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24808896)))]; tensor layers_1_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(24809472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25006144))))[name = string("layers_1_self_attn_k_proj_weight_palettized")]; tensor layers_1_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25006464))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25203136))))[name = string("layers_1_self_attn_v_proj_weight_palettized")]; tensor layers_1_self_attn_k_norm_weight = const()[name = string("layers_1_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25203456)))]; tensor layers_1_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(25204032))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(29922688))))[name = string("layers_1_mlp_gate_proj_weight_palettized")]; tensor layers_1_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(29928896))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34647552))))[name = string("layers_1_mlp_up_proj_weight_palettized")]; tensor layers_1_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(34653760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39372416))))[name = string("layers_1_mlp_down_proj_weight_palettized")]; tensor layers_1_post_feedforward_layernorm_weight = const()[name = string("layers_1_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39374016)))]; tensor layers_1_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39377152))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39573824))))[name = string("layers_1_per_layer_input_gate_weight_palettized")]; tensor layers_2_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(39574144))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41147072))))[name = string("layers_2_self_attn_q_proj_weight_palettized")]; tensor layers_2_self_attn_q_norm_weight = const()[name = string("layers_2_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41149184)))]; tensor layers_2_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41149760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41346432))))[name = string("layers_2_self_attn_k_proj_weight_palettized")]; tensor layers_2_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41346752))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41543424))))[name = string("layers_2_self_attn_v_proj_weight_palettized")]; tensor layers_2_self_attn_k_norm_weight = const()[name = string("layers_2_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41543744)))]; tensor layers_2_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(41544320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46262976))))[name = string("layers_2_mlp_gate_proj_weight_palettized")]; tensor layers_2_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(46269184))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50987840))))[name = string("layers_2_mlp_up_proj_weight_palettized")]; tensor layers_2_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(50994048))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55712704))))[name = string("layers_2_mlp_down_proj_weight_palettized")]; tensor layers_2_post_feedforward_layernorm_weight = const()[name = string("layers_2_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55714304)))]; tensor layers_2_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55717440))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55914112))))[name = string("layers_2_per_layer_input_gate_weight_palettized")]; tensor layers_3_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(55914432))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57487360))))[name = string("layers_3_self_attn_q_proj_weight_palettized")]; tensor layers_3_self_attn_q_norm_weight = const()[name = string("layers_3_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57489472)))]; tensor layers_3_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57490048))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57686720))))[name = string("layers_3_self_attn_k_proj_weight_palettized")]; tensor layers_3_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57687040))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57883712))))[name = string("layers_3_self_attn_v_proj_weight_palettized")]; tensor layers_3_self_attn_k_norm_weight = const()[name = string("layers_3_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57884032)))]; tensor layers_3_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(57884608))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62603264))))[name = string("layers_3_mlp_gate_proj_weight_palettized")]; tensor layers_3_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(62609472))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67328128))))[name = string("layers_3_mlp_up_proj_weight_palettized")]; tensor layers_3_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(67334336))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72052992))))[name = string("layers_3_mlp_down_proj_weight_palettized")]; tensor layers_3_post_feedforward_layernorm_weight = const()[name = string("layers_3_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72054592)))]; tensor layers_3_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72057728))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72254400))))[name = string("layers_3_per_layer_input_gate_weight_palettized")]; tensor layers_4_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(72254720))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75400512))))[name = string("layers_4_self_attn_q_proj_weight_palettized")]; tensor layers_4_self_attn_q_norm_weight = const()[name = string("layers_4_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75404672)))]; tensor layers_4_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75405760))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75799040))))[name = string("layers_4_self_attn_k_proj_weight_palettized")]; tensor layers_4_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(75799616))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(76192896))))[name = string("layers_4_self_attn_v_proj_weight_palettized")]; tensor layers_4_self_attn_k_norm_weight = const()[name = string("layers_4_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(76193472)))]; tensor layers_4_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(76194560))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80913216))))[name = string("layers_4_mlp_gate_proj_weight_palettized")]; tensor layers_4_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(80919424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85638080))))[name = string("layers_4_mlp_up_proj_weight_palettized")]; tensor layers_4_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(85644288))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90362944))))[name = string("layers_4_mlp_down_proj_weight_palettized")]; tensor layers_4_post_feedforward_layernorm_weight = const()[name = string("layers_4_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90364544)))]; tensor layers_4_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90367680))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90564352))))[name = string("layers_4_per_layer_input_gate_weight_palettized")]; tensor layers_5_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90564672))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(92137600))))[name = string("layers_5_self_attn_q_proj_weight_palettized")]; tensor layers_5_self_attn_q_norm_weight = const()[name = string("layers_5_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(92139712)))]; tensor layers_5_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(92140288))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(92336960))))[name = string("layers_5_self_attn_k_proj_weight_palettized")]; tensor layers_5_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(92337280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(92533952))))[name = string("layers_5_self_attn_v_proj_weight_palettized")]; tensor layers_5_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(92534272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(97252928))))[name = string("layers_5_mlp_gate_proj_weight_palettized")]; tensor layers_5_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(97259136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101977792))))[name = string("layers_5_mlp_up_proj_weight_palettized")]; tensor layers_5_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(101984000))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(106702656))))[name = string("layers_5_mlp_down_proj_weight_palettized")]; tensor layers_5_post_feedforward_layernorm_weight = const()[name = string("layers_5_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(106704256)))]; tensor layers_5_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(106707392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(106904064))))[name = string("layers_5_per_layer_input_gate_weight_palettized")]; tensor layers_6_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(106904384))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108477312))))[name = string("layers_6_self_attn_q_proj_weight_palettized")]; tensor layers_6_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108479424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108676096))))[name = string("layers_6_self_attn_k_proj_weight_palettized")]; tensor layers_6_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108676416))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108873088))))[name = string("layers_6_self_attn_v_proj_weight_palettized")]; tensor layers_6_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(108873408))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(113592064))))[name = string("layers_6_mlp_gate_proj_weight_palettized")]; tensor layers_6_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(113598272))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(118316928))))[name = string("layers_6_mlp_up_proj_weight_palettized")]; tensor layers_6_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(118323136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(123041792))))[name = string("layers_6_mlp_down_proj_weight_palettized")]; tensor layers_6_post_feedforward_layernorm_weight = const()[name = string("layers_6_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(123043392)))]; tensor layers_6_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(123046528))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(123243200))))[name = string("layers_6_per_layer_input_gate_weight_palettized")]; tensor layers_7_self_attn_q_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(123243520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(124816448))))[name = string("layers_7_self_attn_q_proj_weight_palettized")]; tensor layers_7_self_attn_q_norm_weight = const()[name = string("layers_7_self_attn_q_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(124818560)))]; tensor layers_7_self_attn_k_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(124819136))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125015808))))[name = string("layers_7_self_attn_k_proj_weight_palettized")]; tensor layers_7_self_attn_v_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125016128))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125212800))))[name = string("layers_7_self_attn_v_proj_weight_palettized")]; tensor layers_7_self_attn_k_norm_weight = const()[name = string("layers_7_self_attn_k_norm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125213120)))]; tensor layers_7_mlp_gate_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(125213696))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(129932352))))[name = string("layers_7_mlp_gate_proj_weight_palettized")]; tensor layers_7_mlp_up_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(129938560))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(134657216))))[name = string("layers_7_mlp_up_proj_weight_palettized")]; tensor layers_7_mlp_down_proj_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(134663424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139382080))))[name = string("layers_7_mlp_down_proj_weight_palettized")]; tensor layers_7_post_feedforward_layernorm_weight = const()[name = string("layers_7_post_feedforward_layernorm_weight"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139383680)))]; tensor layers_7_per_layer_input_gate_weight_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139386816))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139583488))))[name = string("layers_7_per_layer_input_gate_weight_palettized")]; tensor linear_0_bias_0 = const()[name = string("linear_0_bias_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139583808)))]; tensor var_519 = linear(bias = linear_0_bias_0, weight = per_layer_model_projection_weight_palettized, x = hidden_states)[name = string("linear_0")]; fp16 var_520_to_fp16 = const()[name = string("op_520_to_fp16"), val = fp16(0x1.a2p-6)]; tensor proj_cast_fp16 = mul(x = var_519, y = var_520_to_fp16)[name = string("proj_cast_fp16")]; tensor var_526 = const()[name = string("op_526"), val = tensor([1, 8, 35, 256])]; tensor proj_grouped_cast_fp16 = reshape(shape = var_526, x = proj_cast_fp16)[name = string("proj_grouped_cast_fp16")]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_528_cast_fp16 = mul(x = proj_grouped_cast_fp16, y = const_0_promoted_to_fp16)[name = string("op_528_cast_fp16")]; int32 var_530 = const()[name = string("op_530"), val = int32(-1)]; bool input_3_interleave_0 = const()[name = string("input_3_interleave_0"), val = bool(false)]; tensor input_3_cast_fp16 = concat(axis = var_530, interleave = input_3_interleave_0, values = (proj_grouped_cast_fp16, var_528_cast_fp16))[name = string("input_3_cast_fp16")]; tensor normed_1_axes_0 = const()[name = string("normed_1_axes_0"), val = tensor([-1])]; fp16 var_536_to_fp16 = const()[name = string("op_536_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_1_cast_fp16 = layer_norm(axes = normed_1_axes_0, epsilon = var_536_to_fp16, x = input_3_cast_fp16)[name = string("normed_1_cast_fp16")]; tensor var_539_split_sizes_0 = const()[name = string("op_539_split_sizes_0"), val = tensor([256, 256])]; int32 var_539_axis_0 = const()[name = string("op_539_axis_0"), val = int32(-1)]; tensor var_539_cast_fp16_0, tensor var_539_cast_fp16_1 = split(axis = var_539_axis_0, split_sizes = var_539_split_sizes_0, x = normed_1_cast_fp16)[name = string("op_539_cast_fp16")]; tensor per_layer_projection_norm_weight_promoted_to_fp16 = const()[name = string("per_layer_projection_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139601792)))]; tensor var_541_cast_fp16 = mul(x = var_539_cast_fp16_0, y = per_layer_projection_norm_weight_promoted_to_fp16)[name = string("op_541_cast_fp16")]; tensor var_545 = const()[name = string("op_545"), val = tensor([1, 8, 8960])]; tensor proj_normed_cast_fp16 = reshape(shape = var_545, x = var_541_cast_fp16)[name = string("proj_normed_cast_fp16")]; tensor var_548_cast_fp16 = add(x = proj_normed_cast_fp16, y = per_layer_raw)[name = string("op_548_cast_fp16")]; fp16 var_549_to_fp16 = const()[name = string("op_549_to_fp16"), val = fp16(0x1.6ap-1)]; tensor per_layer_combined_out = mul(x = var_548_cast_fp16, y = var_549_to_fp16)[name = string("per_layer_combined_cast_fp16")]; int32 var_555 = const()[name = string("op_555"), val = int32(-1)]; fp16 const_1_promoted = const()[name = string("const_1_promoted"), val = fp16(-0x1p+0)]; tensor var_557 = mul(x = hidden_states, y = const_1_promoted)[name = string("op_557")]; bool input_5_interleave_0 = const()[name = string("input_5_interleave_0"), val = bool(false)]; tensor input_5 = concat(axis = var_555, interleave = input_5_interleave_0, values = (hidden_states, var_557))[name = string("input_5")]; tensor normed_5_axes_0 = const()[name = string("normed_5_axes_0"), val = tensor([-1])]; fp16 var_552_to_fp16 = const()[name = string("op_552_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_5_cast_fp16 = layer_norm(axes = normed_5_axes_0, epsilon = var_552_to_fp16, x = input_5)[name = string("normed_5_cast_fp16")]; tensor var_562_split_sizes_0 = const()[name = string("op_562_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_562_axis_0 = const()[name = string("op_562_axis_0"), val = int32(-1)]; tensor var_562_0, tensor var_562_1 = split(axis = var_562_axis_0, split_sizes = var_562_split_sizes_0, x = normed_5_cast_fp16)[name = string("op_562")]; tensor var_564 = mul(x = var_562_0, y = layers_0_input_layernorm_weight)[name = string("op_564")]; tensor linear_1_bias_0 = const()[name = string("linear_1_bias_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139602368)))]; tensor var_572 = linear(bias = linear_1_bias_0, weight = layers_0_self_attn_q_proj_weight_palettized, x = var_564)[name = string("linear_1")]; tensor var_577 = const()[name = string("op_577"), val = tensor([1, 8, 8, 256])]; tensor var_578 = reshape(shape = var_577, x = var_572)[name = string("op_578")]; tensor var_583 = const()[name = string("op_583"), val = tensor([0, 2, 1, 3])]; int32 var_600 = const()[name = string("op_600"), val = int32(-1)]; fp16 const_2_promoted = const()[name = string("const_2_promoted"), val = fp16(-0x1p+0)]; tensor var_584 = transpose(perm = var_583, x = var_578)[name = string("transpose_79")]; tensor var_602 = mul(x = var_584, y = const_2_promoted)[name = string("op_602")]; bool input_9_interleave_0 = const()[name = string("input_9_interleave_0"), val = bool(false)]; tensor input_9 = concat(axis = var_600, interleave = input_9_interleave_0, values = (var_584, var_602))[name = string("input_9")]; tensor normed_9_axes_0 = const()[name = string("normed_9_axes_0"), val = tensor([-1])]; fp16 var_597_to_fp16 = const()[name = string("op_597_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_9_cast_fp16 = layer_norm(axes = normed_9_axes_0, epsilon = var_597_to_fp16, x = input_9)[name = string("normed_9_cast_fp16")]; tensor var_607_split_sizes_0 = const()[name = string("op_607_split_sizes_0"), val = tensor([256, 256])]; int32 var_607_axis_0 = const()[name = string("op_607_axis_0"), val = int32(-1)]; tensor var_607_0, tensor var_607_1 = split(axis = var_607_axis_0, split_sizes = var_607_split_sizes_0, x = normed_9_cast_fp16)[name = string("op_607")]; tensor q_3 = mul(x = var_607_0, y = layers_0_self_attn_q_norm_weight)[name = string("q_3")]; tensor var_610_cast_fp16 = mul(x = q_3, y = cos_s)[name = string("op_610_cast_fp16")]; tensor var_611_split_sizes_0 = const()[name = string("op_611_split_sizes_0"), val = tensor([128, 128])]; int32 var_611_axis_0 = const()[name = string("op_611_axis_0"), val = int32(-1)]; tensor var_611_0, tensor var_611_1 = split(axis = var_611_axis_0, split_sizes = var_611_split_sizes_0, x = q_3)[name = string("op_611")]; fp16 const_3_promoted = const()[name = string("const_3_promoted"), val = fp16(-0x1p+0)]; tensor var_613 = mul(x = var_611_1, y = const_3_promoted)[name = string("op_613")]; int32 var_615 = const()[name = string("op_615"), val = int32(-1)]; bool var_616_interleave_0 = const()[name = string("op_616_interleave_0"), val = bool(false)]; tensor var_616 = concat(axis = var_615, interleave = var_616_interleave_0, values = (var_613, var_611_0))[name = string("op_616")]; tensor var_617_cast_fp16 = mul(x = var_616, y = sin_s)[name = string("op_617_cast_fp16")]; tensor q_7_cast_fp16 = add(x = var_610_cast_fp16, y = var_617_cast_fp16)[name = string("q_7_cast_fp16")]; tensor linear_2_bias_0 = const()[name = string("linear_2_bias_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139606528)))]; tensor var_622 = linear(bias = linear_2_bias_0, weight = layers_0_self_attn_k_proj_weight_palettized, x = var_564)[name = string("linear_2")]; tensor var_627 = const()[name = string("op_627"), val = tensor([1, 8, 1, 256])]; tensor var_628 = reshape(shape = var_627, x = var_622)[name = string("op_628")]; tensor var_633 = const()[name = string("op_633"), val = tensor([0, 2, 1, 3])]; tensor var_642 = linear(bias = linear_2_bias_0, weight = layers_0_self_attn_v_proj_weight_palettized, x = var_564)[name = string("linear_3")]; tensor var_647 = const()[name = string("op_647"), val = tensor([1, 8, 1, 256])]; tensor var_648 = reshape(shape = var_647, x = var_642)[name = string("op_648")]; tensor var_653 = const()[name = string("op_653"), val = tensor([0, 2, 1, 3])]; int32 var_670 = const()[name = string("op_670"), val = int32(-1)]; fp16 const_4_promoted = const()[name = string("const_4_promoted"), val = fp16(-0x1p+0)]; tensor var_634 = transpose(perm = var_633, x = var_628)[name = string("transpose_78")]; tensor var_672 = mul(x = var_634, y = const_4_promoted)[name = string("op_672")]; bool input_11_interleave_0 = const()[name = string("input_11_interleave_0"), val = bool(false)]; tensor input_11 = concat(axis = var_670, interleave = input_11_interleave_0, values = (var_634, var_672))[name = string("input_11")]; tensor normed_13_axes_0 = const()[name = string("normed_13_axes_0"), val = tensor([-1])]; fp16 var_667_to_fp16 = const()[name = string("op_667_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_13_cast_fp16 = layer_norm(axes = normed_13_axes_0, epsilon = var_667_to_fp16, x = input_11)[name = string("normed_13_cast_fp16")]; tensor var_677_split_sizes_0 = const()[name = string("op_677_split_sizes_0"), val = tensor([256, 256])]; int32 var_677_axis_0 = const()[name = string("op_677_axis_0"), val = int32(-1)]; tensor var_677_0, tensor var_677_1 = split(axis = var_677_axis_0, split_sizes = var_677_split_sizes_0, x = normed_13_cast_fp16)[name = string("op_677")]; tensor q_5 = mul(x = var_677_0, y = layers_0_self_attn_k_norm_weight)[name = string("q_5")]; fp16 var_680_promoted = const()[name = string("op_680_promoted"), val = fp16(0x1p+1)]; tensor var_654 = transpose(perm = var_653, x = var_648)[name = string("transpose_77")]; tensor var_681 = pow(x = var_654, y = var_680_promoted)[name = string("op_681")]; tensor var_686_axes_0 = const()[name = string("op_686_axes_0"), val = tensor([-1])]; bool var_686_keep_dims_0 = const()[name = string("op_686_keep_dims_0"), val = bool(true)]; tensor var_686 = reduce_mean(axes = var_686_axes_0, keep_dims = var_686_keep_dims_0, x = var_681)[name = string("op_686")]; fp16 var_688_to_fp16 = const()[name = string("op_688_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_1_cast_fp16 = add(x = var_686, y = var_688_to_fp16)[name = string("mean_sq_1_cast_fp16")]; fp32 var_690_epsilon_0 = const()[name = string("op_690_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_690_cast_fp16 = rsqrt(epsilon = var_690_epsilon_0, x = mean_sq_1_cast_fp16)[name = string("op_690_cast_fp16")]; tensor input_15_cast_fp16 = mul(x = var_654, y = var_690_cast_fp16)[name = string("input_15_cast_fp16")]; tensor var_692_cast_fp16 = mul(x = q_5, y = cos_s)[name = string("op_692_cast_fp16")]; tensor var_693_split_sizes_0 = const()[name = string("op_693_split_sizes_0"), val = tensor([128, 128])]; int32 var_693_axis_0 = const()[name = string("op_693_axis_0"), val = int32(-1)]; tensor var_693_0, tensor var_693_1 = split(axis = var_693_axis_0, split_sizes = var_693_split_sizes_0, x = q_5)[name = string("op_693")]; fp16 const_5_promoted = const()[name = string("const_5_promoted"), val = fp16(-0x1p+0)]; tensor var_695 = mul(x = var_693_1, y = const_5_promoted)[name = string("op_695")]; int32 var_697 = const()[name = string("op_697"), val = int32(-1)]; bool var_698_interleave_0 = const()[name = string("op_698_interleave_0"), val = bool(false)]; tensor var_698 = concat(axis = var_697, interleave = var_698_interleave_0, values = (var_695, var_693_0))[name = string("op_698")]; tensor var_699_cast_fp16 = mul(x = var_698, y = sin_s)[name = string("op_699_cast_fp16")]; tensor input_13_cast_fp16 = add(x = var_692_cast_fp16, y = var_699_cast_fp16)[name = string("input_13_cast_fp16")]; tensor k_padded_1_pad_0 = const()[name = string("k_padded_1_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_1_mode_0 = const()[name = string("k_padded_1_mode_0"), val = string("constant")]; fp16 const_6_to_fp16 = const()[name = string("const_6_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_1_cast_fp16 = pad(constant_val = const_6_to_fp16, mode = k_padded_1_mode_0, pad = k_padded_1_pad_0, x = input_13_cast_fp16)[name = string("k_padded_1_cast_fp16")]; tensor v_padded_1_pad_0 = const()[name = string("v_padded_1_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_1_mode_0 = const()[name = string("v_padded_1_mode_0"), val = string("constant")]; fp16 const_7_to_fp16 = const()[name = string("const_7_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_1_cast_fp16 = pad(constant_val = const_7_to_fp16, mode = v_padded_1_mode_0, pad = v_padded_1_pad_0, x = input_15_cast_fp16)[name = string("v_padded_1_cast_fp16")]; int32 var_715 = const()[name = string("op_715"), val = int32(8)]; tensor var_716 = add(x = ring_pos, y = var_715)[name = string("op_716")]; tensor read_state_0 = read_state(input = kv_cache_sliding)[name = string("read_state_0")]; tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([0])]; tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([1])]; int32 concat_2_axis_0 = const()[name = string("concat_2_axis_0"), val = int32(0)]; bool concat_2_interleave_0 = const()[name = string("concat_2_interleave_0"), val = bool(false)]; tensor concat_2 = concat(axis = concat_2_axis_0, interleave = concat_2_interleave_0, values = (expand_dims_0, expand_dims_1, ring_pos, expand_dims_3))[name = string("concat_2")]; tensor concat_3_values1_0 = const()[name = string("concat_3_values1_0"), val = tensor([0])]; tensor concat_3_values3_0 = const()[name = string("concat_3_values3_0"), val = tensor([0])]; int32 concat_3_axis_0 = const()[name = string("concat_3_axis_0"), val = int32(0)]; bool concat_3_interleave_0 = const()[name = string("concat_3_interleave_0"), val = bool(false)]; tensor concat_3 = concat(axis = concat_3_axis_0, interleave = concat_3_interleave_0, values = (expand_dims_4, concat_3_values1_0, var_716, concat_3_values3_0))[name = string("concat_3")]; tensor kv_cache_sliding_internal_tensor_assign_1_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_1_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_1_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_2, begin_mask = kv_cache_sliding_internal_tensor_assign_1_begin_mask_0, end = concat_3, end_mask = kv_cache_sliding_internal_tensor_assign_1_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_1_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_1_stride_0, update = k_padded_1_cast_fp16, x = read_state_0)[name = string("kv_cache_sliding_internal_tensor_assign_1_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_1_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_16_write_state")]; tensor coreml_update_state_16 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_16")]; tensor expand_dims_6 = const()[name = string("expand_dims_6"), val = tensor([1])]; tensor expand_dims_7 = const()[name = string("expand_dims_7"), val = tensor([0])]; tensor expand_dims_9 = const()[name = string("expand_dims_9"), val = tensor([0])]; tensor expand_dims_10 = const()[name = string("expand_dims_10"), val = tensor([2])]; int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_6, expand_dims_7, ring_pos, expand_dims_9))[name = string("concat_6")]; tensor concat_7_values1_0 = const()[name = string("concat_7_values1_0"), val = tensor([0])]; tensor concat_7_values3_0 = const()[name = string("concat_7_values3_0"), val = tensor([0])]; int32 concat_7_axis_0 = const()[name = string("concat_7_axis_0"), val = int32(0)]; bool concat_7_interleave_0 = const()[name = string("concat_7_interleave_0"), val = bool(false)]; tensor concat_7 = concat(axis = concat_7_axis_0, interleave = concat_7_interleave_0, values = (expand_dims_10, concat_7_values1_0, var_716, concat_7_values3_0))[name = string("concat_7")]; tensor kv_cache_sliding_internal_tensor_assign_2_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_2_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_2_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_6, begin_mask = kv_cache_sliding_internal_tensor_assign_2_begin_mask_0, end = concat_7, end_mask = kv_cache_sliding_internal_tensor_assign_2_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_2_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_2_stride_0, update = v_padded_1_cast_fp16, x = coreml_update_state_16)[name = string("kv_cache_sliding_internal_tensor_assign_2_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_2_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_17_write_state")]; tensor coreml_update_state_17 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_17")]; tensor var_766_begin_0 = const()[name = string("op_766_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_766_end_0 = const()[name = string("op_766_end_0"), val = tensor([1, 1, 512, 512])]; tensor var_766_end_mask_0 = const()[name = string("op_766_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_766_cast_fp16 = slice_by_index(begin = var_766_begin_0, end = var_766_end_0, end_mask = var_766_end_mask_0, x = coreml_update_state_17)[name = string("op_766_cast_fp16")]; tensor K_sliding_slice_1_begin_0 = const()[name = string("K_sliding_slice_1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_sliding_slice_1_end_0 = const()[name = string("K_sliding_slice_1_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_sliding_slice_1_end_mask_0 = const()[name = string("K_sliding_slice_1_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_sliding_slice_1_cast_fp16 = slice_by_index(begin = K_sliding_slice_1_begin_0, end = K_sliding_slice_1_end_0, end_mask = K_sliding_slice_1_end_mask_0, x = var_766_cast_fp16)[name = string("K_sliding_slice_1_cast_fp16")]; tensor var_786_begin_0 = const()[name = string("op_786_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_786_end_0 = const()[name = string("op_786_end_0"), val = tensor([2, 1, 512, 512])]; tensor var_786_end_mask_0 = const()[name = string("op_786_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_786_cast_fp16 = slice_by_index(begin = var_786_begin_0, end = var_786_end_0, end_mask = var_786_end_mask_0, x = coreml_update_state_17)[name = string("op_786_cast_fp16")]; tensor V_for_attn_1_begin_0 = const()[name = string("V_for_attn_1_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_1_end_0 = const()[name = string("V_for_attn_1_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_1_end_mask_0 = const()[name = string("V_for_attn_1_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_1_cast_fp16 = slice_by_index(begin = V_for_attn_1_begin_0, end = V_for_attn_1_end_0, end_mask = V_for_attn_1_end_mask_0, x = var_786_cast_fp16)[name = string("V_for_attn_1_cast_fp16")]; tensor transpose_0_perm_0 = const()[name = string("transpose_0_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_0_reps_0 = const()[name = string("tile_0_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_0_cast_fp16 = transpose(perm = transpose_0_perm_0, x = K_sliding_slice_1_cast_fp16)[name = string("transpose_76")]; tensor tile_0_cast_fp16 = tile(reps = tile_0_reps_0, x = transpose_0_cast_fp16)[name = string("tile_0_cast_fp16")]; tensor concat_8 = const()[name = string("concat_8"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_0_cast_fp16 = reshape(shape = concat_8, x = tile_0_cast_fp16)[name = string("reshape_0_cast_fp16")]; tensor transpose_1_perm_0 = const()[name = string("transpose_1_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_9 = const()[name = string("concat_9"), val = tensor([-1, 1, 512, 256])]; tensor transpose_1_cast_fp16 = transpose(perm = transpose_1_perm_0, x = reshape_0_cast_fp16)[name = string("transpose_75")]; tensor reshape_1_cast_fp16 = reshape(shape = concat_9, x = transpose_1_cast_fp16)[name = string("reshape_1_cast_fp16")]; tensor transpose_32_perm_0 = const()[name = string("transpose_32_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_2_perm_0 = const()[name = string("transpose_2_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_1_reps_0 = const()[name = string("tile_1_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_2_cast_fp16 = transpose(perm = transpose_2_perm_0, x = V_for_attn_1_cast_fp16)[name = string("transpose_74")]; tensor tile_1_cast_fp16 = tile(reps = tile_1_reps_0, x = transpose_2_cast_fp16)[name = string("tile_1_cast_fp16")]; tensor concat_10 = const()[name = string("concat_10"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_2_cast_fp16 = reshape(shape = concat_10, x = tile_1_cast_fp16)[name = string("reshape_2_cast_fp16")]; tensor transpose_3_perm_0 = const()[name = string("transpose_3_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_11 = const()[name = string("concat_11"), val = tensor([-1, 1, 512, 256])]; tensor transpose_3_cast_fp16 = transpose(perm = transpose_3_perm_0, x = reshape_2_cast_fp16)[name = string("transpose_73")]; tensor reshape_3_cast_fp16 = reshape(shape = concat_11, x = transpose_3_cast_fp16)[name = string("reshape_3_cast_fp16")]; tensor V_expanded_1_perm_0 = const()[name = string("V_expanded_1_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_1_transpose_x_0 = const()[name = string("attn_weights_1_transpose_x_0"), val = bool(false)]; bool attn_weights_1_transpose_y_0 = const()[name = string("attn_weights_1_transpose_y_0"), val = bool(false)]; tensor transpose_32_cast_fp16 = transpose(perm = transpose_32_perm_0, x = reshape_1_cast_fp16)[name = string("transpose_72")]; tensor attn_weights_1_cast_fp16 = matmul(transpose_x = attn_weights_1_transpose_x_0, transpose_y = attn_weights_1_transpose_y_0, x = q_7_cast_fp16, y = transpose_32_cast_fp16)[name = string("attn_weights_1_cast_fp16")]; tensor x_7_cast_fp16 = add(x = attn_weights_1_cast_fp16, y = causal_mask_sliding)[name = string("x_7_cast_fp16")]; tensor reduce_max_0_axes_0 = const()[name = string("reduce_max_0_axes_0"), val = tensor([-1])]; bool reduce_max_0_keep_dims_0 = const()[name = string("reduce_max_0_keep_dims_0"), val = bool(true)]; tensor reduce_max_0 = reduce_max(axes = reduce_max_0_axes_0, keep_dims = reduce_max_0_keep_dims_0, x = x_7_cast_fp16)[name = string("reduce_max_0")]; tensor var_831 = sub(x = x_7_cast_fp16, y = reduce_max_0)[name = string("op_831")]; tensor var_837 = exp(x = var_831)[name = string("op_837")]; tensor var_847_axes_0 = const()[name = string("op_847_axes_0"), val = tensor([-1])]; bool var_847_keep_dims_0 = const()[name = string("op_847_keep_dims_0"), val = bool(true)]; tensor var_847 = reduce_sum(axes = var_847_axes_0, keep_dims = var_847_keep_dims_0, x = var_837)[name = string("op_847")]; tensor var_853_cast_fp16 = real_div(x = var_837, y = var_847)[name = string("op_853_cast_fp16")]; bool attn_output_1_transpose_x_0 = const()[name = string("attn_output_1_transpose_x_0"), val = bool(false)]; bool attn_output_1_transpose_y_0 = const()[name = string("attn_output_1_transpose_y_0"), val = bool(false)]; tensor V_expanded_1_cast_fp16 = transpose(perm = V_expanded_1_perm_0, x = reshape_3_cast_fp16)[name = string("transpose_71")]; tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_0, transpose_y = attn_output_1_transpose_y_0, x = var_853_cast_fp16, y = V_expanded_1_cast_fp16)[name = string("attn_output_1_cast_fp16")]; tensor var_864 = const()[name = string("op_864"), val = tensor([0, 2, 1, 3])]; tensor var_871 = const()[name = string("op_871"), val = tensor([1, 8, -1])]; tensor var_865_cast_fp16 = transpose(perm = var_864, x = attn_output_1_cast_fp16)[name = string("transpose_70")]; tensor input_17_cast_fp16 = reshape(shape = var_871, x = var_865_cast_fp16)[name = string("input_17_cast_fp16")]; tensor layers_0_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(139607104))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141180032))))[name = string("layers_0_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_4_bias_0_to_fp16 = const()[name = string("linear_4_bias_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141181632)))]; tensor linear_4_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_0_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_17_cast_fp16)[name = string("linear_4_cast_fp16")]; int32 var_880 = const()[name = string("op_880"), val = int32(-1)]; fp16 const_8_promoted_to_fp16 = const()[name = string("const_8_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_882_cast_fp16 = mul(x = linear_4_cast_fp16, y = const_8_promoted_to_fp16)[name = string("op_882_cast_fp16")]; bool input_19_interleave_0 = const()[name = string("input_19_interleave_0"), val = bool(false)]; tensor input_19_cast_fp16 = concat(axis = var_880, interleave = input_19_interleave_0, values = (linear_4_cast_fp16, var_882_cast_fp16))[name = string("input_19_cast_fp16")]; tensor normed_17_axes_0 = const()[name = string("normed_17_axes_0"), val = tensor([-1])]; fp16 var_877_to_fp16 = const()[name = string("op_877_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_17_cast_fp16 = layer_norm(axes = normed_17_axes_0, epsilon = var_877_to_fp16, x = input_19_cast_fp16)[name = string("normed_17_cast_fp16")]; tensor var_887_split_sizes_0 = const()[name = string("op_887_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_887_axis_0 = const()[name = string("op_887_axis_0"), val = int32(-1)]; tensor var_887_cast_fp16_0, tensor var_887_cast_fp16_1 = split(axis = var_887_axis_0, split_sizes = var_887_split_sizes_0, x = normed_17_cast_fp16)[name = string("op_887_cast_fp16")]; tensor layers_0_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_0_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141184768)))]; tensor attn_output_3_cast_fp16 = mul(x = var_887_cast_fp16_0, y = layers_0_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_3_cast_fp16")]; tensor x_13_cast_fp16 = add(x = hidden_states, y = attn_output_3_cast_fp16)[name = string("x_13_cast_fp16")]; int32 var_896 = const()[name = string("op_896"), val = int32(-1)]; fp16 const_9_promoted_to_fp16 = const()[name = string("const_9_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_898_cast_fp16 = mul(x = x_13_cast_fp16, y = const_9_promoted_to_fp16)[name = string("op_898_cast_fp16")]; bool input_21_interleave_0 = const()[name = string("input_21_interleave_0"), val = bool(false)]; tensor input_21_cast_fp16 = concat(axis = var_896, interleave = input_21_interleave_0, values = (x_13_cast_fp16, var_898_cast_fp16))[name = string("input_21_cast_fp16")]; tensor normed_21_axes_0 = const()[name = string("normed_21_axes_0"), val = tensor([-1])]; fp16 var_893_to_fp16 = const()[name = string("op_893_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_21_cast_fp16 = layer_norm(axes = normed_21_axes_0, epsilon = var_893_to_fp16, x = input_21_cast_fp16)[name = string("normed_21_cast_fp16")]; tensor var_903_split_sizes_0 = const()[name = string("op_903_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_903_axis_0 = const()[name = string("op_903_axis_0"), val = int32(-1)]; tensor var_903_cast_fp16_0, tensor var_903_cast_fp16_1 = split(axis = var_903_axis_0, split_sizes = var_903_split_sizes_0, x = normed_21_cast_fp16)[name = string("op_903_cast_fp16")]; tensor layers_0_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_0_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141187904)))]; tensor var_905_cast_fp16 = mul(x = var_903_cast_fp16_0, y = layers_0_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("op_905_cast_fp16")]; tensor linear_5_bias_0 = const()[name = string("linear_5_bias_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141191040)))]; tensor gate_1 = linear(bias = linear_5_bias_0, weight = layers_0_mlp_gate_proj_weight_palettized, x = var_905_cast_fp16)[name = string("linear_5")]; tensor up_1 = linear(bias = linear_5_bias_0, weight = layers_0_mlp_up_proj_weight_palettized, x = var_905_cast_fp16)[name = string("linear_6")]; string gate_3_mode_0 = const()[name = string("gate_3_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_3 = gelu(mode = gate_3_mode_0, x = gate_1)[name = string("gate_3")]; tensor input_25 = mul(x = gate_3, y = up_1)[name = string("input_25")]; tensor x_15 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_0_mlp_down_proj_weight_palettized, x = input_25)[name = string("linear_7")]; int32 var_927 = const()[name = string("op_927"), val = int32(-1)]; fp16 const_10_promoted = const()[name = string("const_10_promoted"), val = fp16(-0x1p+0)]; tensor var_929 = mul(x = x_15, y = const_10_promoted)[name = string("op_929")]; bool input_27_interleave_0 = const()[name = string("input_27_interleave_0"), val = bool(false)]; tensor input_27 = concat(axis = var_927, interleave = input_27_interleave_0, values = (x_15, var_929))[name = string("input_27")]; tensor normed_25_axes_0 = const()[name = string("normed_25_axes_0"), val = tensor([-1])]; fp16 var_924_to_fp16 = const()[name = string("op_924_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_25_cast_fp16 = layer_norm(axes = normed_25_axes_0, epsilon = var_924_to_fp16, x = input_27)[name = string("normed_25_cast_fp16")]; tensor var_934_split_sizes_0 = const()[name = string("op_934_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_934_axis_0 = const()[name = string("op_934_axis_0"), val = int32(-1)]; tensor var_934_0, tensor var_934_1 = split(axis = var_934_axis_0, split_sizes = var_934_split_sizes_0, x = normed_25_cast_fp16)[name = string("op_934")]; tensor hidden_states_3 = mul(x = var_934_0, y = layers_0_post_feedforward_layernorm_weight)[name = string("hidden_states_3")]; tensor hidden_states_5_cast_fp16 = add(x = x_13_cast_fp16, y = hidden_states_3)[name = string("hidden_states_5_cast_fp16")]; tensor per_layer_slice_1_begin_0 = const()[name = string("per_layer_slice_1_begin_0"), val = tensor([0, 0, 0])]; tensor per_layer_slice_1_end_0 = const()[name = string("per_layer_slice_1_end_0"), val = tensor([1, 8, 256])]; tensor per_layer_slice_1_end_mask_0 = const()[name = string("per_layer_slice_1_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_1_cast_fp16 = slice_by_index(begin = per_layer_slice_1_begin_0, end = per_layer_slice_1_end_0, end_mask = per_layer_slice_1_end_mask_0, x = per_layer_combined_out)[name = string("per_layer_slice_1_cast_fp16")]; tensor gated_1 = linear(bias = linear_2_bias_0, weight = layers_0_per_layer_input_gate_weight_palettized, x = hidden_states_5_cast_fp16)[name = string("linear_8")]; string gated_3_mode_0 = const()[name = string("gated_3_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_3 = gelu(mode = gated_3_mode_0, x = gated_1)[name = string("gated_3")]; tensor input_31_cast_fp16 = mul(x = gated_3, y = per_layer_slice_1_cast_fp16)[name = string("input_31_cast_fp16")]; tensor layers_0_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141203392))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141400064))))[name = string("layers_0_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_9_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_0_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_31_cast_fp16)[name = string("linear_9_cast_fp16")]; int32 var_972 = const()[name = string("op_972"), val = int32(-1)]; fp16 const_11_promoted_to_fp16 = const()[name = string("const_11_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_974_cast_fp16 = mul(x = linear_9_cast_fp16, y = const_11_promoted_to_fp16)[name = string("op_974_cast_fp16")]; bool input_33_interleave_0 = const()[name = string("input_33_interleave_0"), val = bool(false)]; tensor input_33_cast_fp16 = concat(axis = var_972, interleave = input_33_interleave_0, values = (linear_9_cast_fp16, var_974_cast_fp16))[name = string("input_33_cast_fp16")]; tensor normed_29_axes_0 = const()[name = string("normed_29_axes_0"), val = tensor([-1])]; fp16 var_969_to_fp16 = const()[name = string("op_969_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_29_cast_fp16 = layer_norm(axes = normed_29_axes_0, epsilon = var_969_to_fp16, x = input_33_cast_fp16)[name = string("normed_29_cast_fp16")]; tensor var_979_split_sizes_0 = const()[name = string("op_979_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_979_axis_0 = const()[name = string("op_979_axis_0"), val = int32(-1)]; tensor var_979_cast_fp16_0, tensor var_979_cast_fp16_1 = split(axis = var_979_axis_0, split_sizes = var_979_split_sizes_0, x = normed_29_cast_fp16)[name = string("op_979_cast_fp16")]; tensor layers_0_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_0_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141401664)))]; tensor hidden_states_7_cast_fp16 = mul(x = var_979_cast_fp16_0, y = layers_0_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_7_cast_fp16")]; tensor hidden_states_9_cast_fp16 = add(x = hidden_states_5_cast_fp16, y = hidden_states_7_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; tensor const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = tensor([0x1.24p-6])]; tensor x_19_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_12_promoted_to_fp16)[name = string("x_19_cast_fp16")]; int32 var_994 = const()[name = string("op_994"), val = int32(-1)]; fp16 const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_996_cast_fp16 = mul(x = x_19_cast_fp16, y = const_13_promoted_to_fp16)[name = string("op_996_cast_fp16")]; bool input_35_interleave_0 = const()[name = string("input_35_interleave_0"), val = bool(false)]; tensor input_35_cast_fp16 = concat(axis = var_994, interleave = input_35_interleave_0, values = (x_19_cast_fp16, var_996_cast_fp16))[name = string("input_35_cast_fp16")]; tensor normed_33_axes_0 = const()[name = string("normed_33_axes_0"), val = tensor([-1])]; fp16 var_991_to_fp16 = const()[name = string("op_991_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_33_cast_fp16 = layer_norm(axes = normed_33_axes_0, epsilon = var_991_to_fp16, x = input_35_cast_fp16)[name = string("normed_33_cast_fp16")]; tensor var_1001_split_sizes_0 = const()[name = string("op_1001_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1001_axis_0 = const()[name = string("op_1001_axis_0"), val = int32(-1)]; tensor var_1001_cast_fp16_0, tensor var_1001_cast_fp16_1 = split(axis = var_1001_axis_0, split_sizes = var_1001_split_sizes_0, x = normed_33_cast_fp16)[name = string("op_1001_cast_fp16")]; tensor layers_1_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_1_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141404800)))]; tensor var_1003_cast_fp16 = mul(x = var_1001_cast_fp16_0, y = layers_1_input_layernorm_weight_promoted_to_fp16)[name = string("op_1003_cast_fp16")]; tensor var_1011 = linear(bias = linear_1_bias_0, weight = layers_1_self_attn_q_proj_weight_palettized, x = var_1003_cast_fp16)[name = string("linear_10")]; tensor var_1016 = const()[name = string("op_1016"), val = tensor([1, 8, 8, 256])]; tensor var_1017 = reshape(shape = var_1016, x = var_1011)[name = string("op_1017")]; tensor var_1022 = const()[name = string("op_1022"), val = tensor([0, 2, 1, 3])]; int32 var_1039 = const()[name = string("op_1039"), val = int32(-1)]; fp16 const_14_promoted = const()[name = string("const_14_promoted"), val = fp16(-0x1p+0)]; tensor var_1023 = transpose(perm = var_1022, x = var_1017)[name = string("transpose_69")]; tensor var_1041 = mul(x = var_1023, y = const_14_promoted)[name = string("op_1041")]; bool input_39_interleave_0 = const()[name = string("input_39_interleave_0"), val = bool(false)]; tensor input_39 = concat(axis = var_1039, interleave = input_39_interleave_0, values = (var_1023, var_1041))[name = string("input_39")]; tensor normed_37_axes_0 = const()[name = string("normed_37_axes_0"), val = tensor([-1])]; fp16 var_1036_to_fp16 = const()[name = string("op_1036_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_37_cast_fp16 = layer_norm(axes = normed_37_axes_0, epsilon = var_1036_to_fp16, x = input_39)[name = string("normed_37_cast_fp16")]; tensor var_1046_split_sizes_0 = const()[name = string("op_1046_split_sizes_0"), val = tensor([256, 256])]; int32 var_1046_axis_0 = const()[name = string("op_1046_axis_0"), val = int32(-1)]; tensor var_1046_0, tensor var_1046_1 = split(axis = var_1046_axis_0, split_sizes = var_1046_split_sizes_0, x = normed_37_cast_fp16)[name = string("op_1046")]; tensor q_11 = mul(x = var_1046_0, y = layers_1_self_attn_q_norm_weight)[name = string("q_11")]; tensor var_1049_cast_fp16 = mul(x = q_11, y = cos_s)[name = string("op_1049_cast_fp16")]; tensor var_1050_split_sizes_0 = const()[name = string("op_1050_split_sizes_0"), val = tensor([128, 128])]; int32 var_1050_axis_0 = const()[name = string("op_1050_axis_0"), val = int32(-1)]; tensor var_1050_0, tensor var_1050_1 = split(axis = var_1050_axis_0, split_sizes = var_1050_split_sizes_0, x = q_11)[name = string("op_1050")]; fp16 const_15_promoted = const()[name = string("const_15_promoted"), val = fp16(-0x1p+0)]; tensor var_1052 = mul(x = var_1050_1, y = const_15_promoted)[name = string("op_1052")]; int32 var_1054 = const()[name = string("op_1054"), val = int32(-1)]; bool var_1055_interleave_0 = const()[name = string("op_1055_interleave_0"), val = bool(false)]; tensor var_1055 = concat(axis = var_1054, interleave = var_1055_interleave_0, values = (var_1052, var_1050_0))[name = string("op_1055")]; tensor var_1056_cast_fp16 = mul(x = var_1055, y = sin_s)[name = string("op_1056_cast_fp16")]; tensor q_15_cast_fp16 = add(x = var_1049_cast_fp16, y = var_1056_cast_fp16)[name = string("q_15_cast_fp16")]; tensor var_1061 = linear(bias = linear_2_bias_0, weight = layers_1_self_attn_k_proj_weight_palettized, x = var_1003_cast_fp16)[name = string("linear_11")]; tensor var_1066 = const()[name = string("op_1066"), val = tensor([1, 8, 1, 256])]; tensor var_1067 = reshape(shape = var_1066, x = var_1061)[name = string("op_1067")]; tensor var_1072 = const()[name = string("op_1072"), val = tensor([0, 2, 1, 3])]; tensor var_1081 = linear(bias = linear_2_bias_0, weight = layers_1_self_attn_v_proj_weight_palettized, x = var_1003_cast_fp16)[name = string("linear_12")]; tensor var_1086 = const()[name = string("op_1086"), val = tensor([1, 8, 1, 256])]; tensor var_1087 = reshape(shape = var_1086, x = var_1081)[name = string("op_1087")]; tensor var_1092 = const()[name = string("op_1092"), val = tensor([0, 2, 1, 3])]; int32 var_1109 = const()[name = string("op_1109"), val = int32(-1)]; fp16 const_16_promoted = const()[name = string("const_16_promoted"), val = fp16(-0x1p+0)]; tensor var_1073 = transpose(perm = var_1072, x = var_1067)[name = string("transpose_68")]; tensor var_1111 = mul(x = var_1073, y = const_16_promoted)[name = string("op_1111")]; bool input_41_interleave_0 = const()[name = string("input_41_interleave_0"), val = bool(false)]; tensor input_41 = concat(axis = var_1109, interleave = input_41_interleave_0, values = (var_1073, var_1111))[name = string("input_41")]; tensor normed_41_axes_0 = const()[name = string("normed_41_axes_0"), val = tensor([-1])]; fp16 var_1106_to_fp16 = const()[name = string("op_1106_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_41_cast_fp16 = layer_norm(axes = normed_41_axes_0, epsilon = var_1106_to_fp16, x = input_41)[name = string("normed_41_cast_fp16")]; tensor var_1116_split_sizes_0 = const()[name = string("op_1116_split_sizes_0"), val = tensor([256, 256])]; int32 var_1116_axis_0 = const()[name = string("op_1116_axis_0"), val = int32(-1)]; tensor var_1116_0, tensor var_1116_1 = split(axis = var_1116_axis_0, split_sizes = var_1116_split_sizes_0, x = normed_41_cast_fp16)[name = string("op_1116")]; tensor q_13 = mul(x = var_1116_0, y = layers_1_self_attn_k_norm_weight)[name = string("q_13")]; fp16 var_1119_promoted = const()[name = string("op_1119_promoted"), val = fp16(0x1p+1)]; tensor var_1093 = transpose(perm = var_1092, x = var_1087)[name = string("transpose_67")]; tensor var_1120 = pow(x = var_1093, y = var_1119_promoted)[name = string("op_1120")]; tensor var_1125_axes_0 = const()[name = string("op_1125_axes_0"), val = tensor([-1])]; bool var_1125_keep_dims_0 = const()[name = string("op_1125_keep_dims_0"), val = bool(true)]; tensor var_1125 = reduce_mean(axes = var_1125_axes_0, keep_dims = var_1125_keep_dims_0, x = var_1120)[name = string("op_1125")]; fp16 var_1127_to_fp16 = const()[name = string("op_1127_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_3_cast_fp16 = add(x = var_1125, y = var_1127_to_fp16)[name = string("mean_sq_3_cast_fp16")]; fp32 var_1129_epsilon_0 = const()[name = string("op_1129_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_1129_cast_fp16 = rsqrt(epsilon = var_1129_epsilon_0, x = mean_sq_3_cast_fp16)[name = string("op_1129_cast_fp16")]; tensor input_45_cast_fp16 = mul(x = var_1093, y = var_1129_cast_fp16)[name = string("input_45_cast_fp16")]; tensor var_1131_cast_fp16 = mul(x = q_13, y = cos_s)[name = string("op_1131_cast_fp16")]; tensor var_1132_split_sizes_0 = const()[name = string("op_1132_split_sizes_0"), val = tensor([128, 128])]; int32 var_1132_axis_0 = const()[name = string("op_1132_axis_0"), val = int32(-1)]; tensor var_1132_0, tensor var_1132_1 = split(axis = var_1132_axis_0, split_sizes = var_1132_split_sizes_0, x = q_13)[name = string("op_1132")]; fp16 const_17_promoted = const()[name = string("const_17_promoted"), val = fp16(-0x1p+0)]; tensor var_1134 = mul(x = var_1132_1, y = const_17_promoted)[name = string("op_1134")]; int32 var_1136 = const()[name = string("op_1136"), val = int32(-1)]; bool var_1137_interleave_0 = const()[name = string("op_1137_interleave_0"), val = bool(false)]; tensor var_1137 = concat(axis = var_1136, interleave = var_1137_interleave_0, values = (var_1134, var_1132_0))[name = string("op_1137")]; tensor var_1138_cast_fp16 = mul(x = var_1137, y = sin_s)[name = string("op_1138_cast_fp16")]; tensor input_43_cast_fp16 = add(x = var_1131_cast_fp16, y = var_1138_cast_fp16)[name = string("input_43_cast_fp16")]; tensor k_padded_3_pad_0 = const()[name = string("k_padded_3_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_3_mode_0 = const()[name = string("k_padded_3_mode_0"), val = string("constant")]; fp16 const_18_to_fp16 = const()[name = string("const_18_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_3_cast_fp16 = pad(constant_val = const_18_to_fp16, mode = k_padded_3_mode_0, pad = k_padded_3_pad_0, x = input_43_cast_fp16)[name = string("k_padded_3_cast_fp16")]; tensor v_padded_3_pad_0 = const()[name = string("v_padded_3_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_3_mode_0 = const()[name = string("v_padded_3_mode_0"), val = string("constant")]; fp16 const_19_to_fp16 = const()[name = string("const_19_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_3_cast_fp16 = pad(constant_val = const_19_to_fp16, mode = v_padded_3_mode_0, pad = v_padded_3_pad_0, x = input_45_cast_fp16)[name = string("v_padded_3_cast_fp16")]; tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([2])]; tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([3])]; int32 concat_14_axis_0 = const()[name = string("concat_14_axis_0"), val = int32(0)]; bool concat_14_interleave_0 = const()[name = string("concat_14_interleave_0"), val = bool(false)]; tensor concat_14 = concat(axis = concat_14_axis_0, interleave = concat_14_interleave_0, values = (expand_dims_12, expand_dims_13, ring_pos, expand_dims_15))[name = string("concat_14")]; tensor concat_15_values1_0 = const()[name = string("concat_15_values1_0"), val = tensor([0])]; tensor concat_15_values3_0 = const()[name = string("concat_15_values3_0"), val = tensor([0])]; int32 concat_15_axis_0 = const()[name = string("concat_15_axis_0"), val = int32(0)]; bool concat_15_interleave_0 = const()[name = string("concat_15_interleave_0"), val = bool(false)]; tensor concat_15 = concat(axis = concat_15_axis_0, interleave = concat_15_interleave_0, values = (expand_dims_16, concat_15_values1_0, var_716, concat_15_values3_0))[name = string("concat_15")]; tensor kv_cache_sliding_internal_tensor_assign_3_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_3_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_3_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_14, begin_mask = kv_cache_sliding_internal_tensor_assign_3_begin_mask_0, end = concat_15, end_mask = kv_cache_sliding_internal_tensor_assign_3_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_3_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_3_stride_0, update = k_padded_3_cast_fp16, x = coreml_update_state_17)[name = string("kv_cache_sliding_internal_tensor_assign_3_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_3_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_18_write_state")]; tensor coreml_update_state_18 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_18")]; tensor expand_dims_18 = const()[name = string("expand_dims_18"), val = tensor([3])]; tensor expand_dims_19 = const()[name = string("expand_dims_19"), val = tensor([0])]; tensor expand_dims_21 = const()[name = string("expand_dims_21"), val = tensor([0])]; tensor expand_dims_22 = const()[name = string("expand_dims_22"), val = tensor([4])]; int32 concat_18_axis_0 = const()[name = string("concat_18_axis_0"), val = int32(0)]; bool concat_18_interleave_0 = const()[name = string("concat_18_interleave_0"), val = bool(false)]; tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (expand_dims_18, expand_dims_19, ring_pos, expand_dims_21))[name = string("concat_18")]; tensor concat_19_values1_0 = const()[name = string("concat_19_values1_0"), val = tensor([0])]; tensor concat_19_values3_0 = const()[name = string("concat_19_values3_0"), val = tensor([0])]; int32 concat_19_axis_0 = const()[name = string("concat_19_axis_0"), val = int32(0)]; bool concat_19_interleave_0 = const()[name = string("concat_19_interleave_0"), val = bool(false)]; tensor concat_19 = concat(axis = concat_19_axis_0, interleave = concat_19_interleave_0, values = (expand_dims_22, concat_19_values1_0, var_716, concat_19_values3_0))[name = string("concat_19")]; tensor kv_cache_sliding_internal_tensor_assign_4_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_4_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_4_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_18, begin_mask = kv_cache_sliding_internal_tensor_assign_4_begin_mask_0, end = concat_19, end_mask = kv_cache_sliding_internal_tensor_assign_4_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_4_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_4_stride_0, update = v_padded_3_cast_fp16, x = coreml_update_state_18)[name = string("kv_cache_sliding_internal_tensor_assign_4_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_4_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_19_write_state")]; tensor coreml_update_state_19 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_19")]; tensor var_1205_begin_0 = const()[name = string("op_1205_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1205_end_0 = const()[name = string("op_1205_end_0"), val = tensor([3, 1, 512, 512])]; tensor var_1205_end_mask_0 = const()[name = string("op_1205_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1205_cast_fp16 = slice_by_index(begin = var_1205_begin_0, end = var_1205_end_0, end_mask = var_1205_end_mask_0, x = coreml_update_state_19)[name = string("op_1205_cast_fp16")]; tensor K_sliding_slice_3_begin_0 = const()[name = string("K_sliding_slice_3_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_sliding_slice_3_end_0 = const()[name = string("K_sliding_slice_3_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_sliding_slice_3_end_mask_0 = const()[name = string("K_sliding_slice_3_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_sliding_slice_3_cast_fp16 = slice_by_index(begin = K_sliding_slice_3_begin_0, end = K_sliding_slice_3_end_0, end_mask = K_sliding_slice_3_end_mask_0, x = var_1205_cast_fp16)[name = string("K_sliding_slice_3_cast_fp16")]; tensor var_1225_begin_0 = const()[name = string("op_1225_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_1225_end_0 = const()[name = string("op_1225_end_0"), val = tensor([4, 1, 512, 512])]; tensor var_1225_end_mask_0 = const()[name = string("op_1225_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1225_cast_fp16 = slice_by_index(begin = var_1225_begin_0, end = var_1225_end_0, end_mask = var_1225_end_mask_0, x = coreml_update_state_19)[name = string("op_1225_cast_fp16")]; tensor V_for_attn_3_begin_0 = const()[name = string("V_for_attn_3_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_3_end_0 = const()[name = string("V_for_attn_3_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_3_end_mask_0 = const()[name = string("V_for_attn_3_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_3_cast_fp16 = slice_by_index(begin = V_for_attn_3_begin_0, end = V_for_attn_3_end_0, end_mask = V_for_attn_3_end_mask_0, x = var_1225_cast_fp16)[name = string("V_for_attn_3_cast_fp16")]; tensor transpose_4_perm_0 = const()[name = string("transpose_4_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_2_reps_0 = const()[name = string("tile_2_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_4_cast_fp16 = transpose(perm = transpose_4_perm_0, x = K_sliding_slice_3_cast_fp16)[name = string("transpose_66")]; tensor tile_2_cast_fp16 = tile(reps = tile_2_reps_0, x = transpose_4_cast_fp16)[name = string("tile_2_cast_fp16")]; tensor concat_20 = const()[name = string("concat_20"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_4_cast_fp16 = reshape(shape = concat_20, x = tile_2_cast_fp16)[name = string("reshape_4_cast_fp16")]; tensor transpose_5_perm_0 = const()[name = string("transpose_5_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_21 = const()[name = string("concat_21"), val = tensor([-1, 1, 512, 256])]; tensor transpose_5_cast_fp16 = transpose(perm = transpose_5_perm_0, x = reshape_4_cast_fp16)[name = string("transpose_65")]; tensor reshape_5_cast_fp16 = reshape(shape = concat_21, x = transpose_5_cast_fp16)[name = string("reshape_5_cast_fp16")]; tensor transpose_33_perm_0 = const()[name = string("transpose_33_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_6_perm_0 = const()[name = string("transpose_6_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_3_reps_0 = const()[name = string("tile_3_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_6_cast_fp16 = transpose(perm = transpose_6_perm_0, x = V_for_attn_3_cast_fp16)[name = string("transpose_64")]; tensor tile_3_cast_fp16 = tile(reps = tile_3_reps_0, x = transpose_6_cast_fp16)[name = string("tile_3_cast_fp16")]; tensor concat_22 = const()[name = string("concat_22"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_6_cast_fp16 = reshape(shape = concat_22, x = tile_3_cast_fp16)[name = string("reshape_6_cast_fp16")]; tensor transpose_7_perm_0 = const()[name = string("transpose_7_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_23 = const()[name = string("concat_23"), val = tensor([-1, 1, 512, 256])]; tensor transpose_7_cast_fp16 = transpose(perm = transpose_7_perm_0, x = reshape_6_cast_fp16)[name = string("transpose_63")]; tensor reshape_7_cast_fp16 = reshape(shape = concat_23, x = transpose_7_cast_fp16)[name = string("reshape_7_cast_fp16")]; tensor V_expanded_3_perm_0 = const()[name = string("V_expanded_3_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_5_transpose_x_0 = const()[name = string("attn_weights_5_transpose_x_0"), val = bool(false)]; bool attn_weights_5_transpose_y_0 = const()[name = string("attn_weights_5_transpose_y_0"), val = bool(false)]; tensor transpose_33_cast_fp16 = transpose(perm = transpose_33_perm_0, x = reshape_5_cast_fp16)[name = string("transpose_62")]; tensor attn_weights_5_cast_fp16 = matmul(transpose_x = attn_weights_5_transpose_x_0, transpose_y = attn_weights_5_transpose_y_0, x = q_15_cast_fp16, y = transpose_33_cast_fp16)[name = string("attn_weights_5_cast_fp16")]; tensor x_27_cast_fp16 = add(x = attn_weights_5_cast_fp16, y = causal_mask_sliding)[name = string("x_27_cast_fp16")]; tensor reduce_max_1_axes_0 = const()[name = string("reduce_max_1_axes_0"), val = tensor([-1])]; bool reduce_max_1_keep_dims_0 = const()[name = string("reduce_max_1_keep_dims_0"), val = bool(true)]; tensor reduce_max_1 = reduce_max(axes = reduce_max_1_axes_0, keep_dims = reduce_max_1_keep_dims_0, x = x_27_cast_fp16)[name = string("reduce_max_1")]; tensor var_1270 = sub(x = x_27_cast_fp16, y = reduce_max_1)[name = string("op_1270")]; tensor var_1276 = exp(x = var_1270)[name = string("op_1276")]; tensor var_1286_axes_0 = const()[name = string("op_1286_axes_0"), val = tensor([-1])]; bool var_1286_keep_dims_0 = const()[name = string("op_1286_keep_dims_0"), val = bool(true)]; tensor var_1286 = reduce_sum(axes = var_1286_axes_0, keep_dims = var_1286_keep_dims_0, x = var_1276)[name = string("op_1286")]; tensor var_1292_cast_fp16 = real_div(x = var_1276, y = var_1286)[name = string("op_1292_cast_fp16")]; bool attn_output_5_transpose_x_0 = const()[name = string("attn_output_5_transpose_x_0"), val = bool(false)]; bool attn_output_5_transpose_y_0 = const()[name = string("attn_output_5_transpose_y_0"), val = bool(false)]; tensor V_expanded_3_cast_fp16 = transpose(perm = V_expanded_3_perm_0, x = reshape_7_cast_fp16)[name = string("transpose_61")]; tensor attn_output_5_cast_fp16 = matmul(transpose_x = attn_output_5_transpose_x_0, transpose_y = attn_output_5_transpose_y_0, x = var_1292_cast_fp16, y = V_expanded_3_cast_fp16)[name = string("attn_output_5_cast_fp16")]; tensor var_1303 = const()[name = string("op_1303"), val = tensor([0, 2, 1, 3])]; tensor var_1310 = const()[name = string("op_1310"), val = tensor([1, 8, -1])]; tensor var_1304_cast_fp16 = transpose(perm = var_1303, x = attn_output_5_cast_fp16)[name = string("transpose_60")]; tensor input_47_cast_fp16 = reshape(shape = var_1310, x = var_1304_cast_fp16)[name = string("input_47_cast_fp16")]; tensor layers_1_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(141407936))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(142980864))))[name = string("layers_1_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_13_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_1_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_47_cast_fp16)[name = string("linear_13_cast_fp16")]; int32 var_1319 = const()[name = string("op_1319"), val = int32(-1)]; fp16 const_20_promoted_to_fp16 = const()[name = string("const_20_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1321_cast_fp16 = mul(x = linear_13_cast_fp16, y = const_20_promoted_to_fp16)[name = string("op_1321_cast_fp16")]; bool input_49_interleave_0 = const()[name = string("input_49_interleave_0"), val = bool(false)]; tensor input_49_cast_fp16 = concat(axis = var_1319, interleave = input_49_interleave_0, values = (linear_13_cast_fp16, var_1321_cast_fp16))[name = string("input_49_cast_fp16")]; tensor normed_45_axes_0 = const()[name = string("normed_45_axes_0"), val = tensor([-1])]; fp16 var_1316_to_fp16 = const()[name = string("op_1316_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_45_cast_fp16 = layer_norm(axes = normed_45_axes_0, epsilon = var_1316_to_fp16, x = input_49_cast_fp16)[name = string("normed_45_cast_fp16")]; tensor var_1326_split_sizes_0 = const()[name = string("op_1326_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1326_axis_0 = const()[name = string("op_1326_axis_0"), val = int32(-1)]; tensor var_1326_cast_fp16_0, tensor var_1326_cast_fp16_1 = split(axis = var_1326_axis_0, split_sizes = var_1326_split_sizes_0, x = normed_45_cast_fp16)[name = string("op_1326_cast_fp16")]; tensor layers_1_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_1_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(142982464)))]; tensor attn_output_7_cast_fp16 = mul(x = var_1326_cast_fp16_0, y = layers_1_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_7_cast_fp16")]; tensor x_33_cast_fp16 = add(x = x_19_cast_fp16, y = attn_output_7_cast_fp16)[name = string("x_33_cast_fp16")]; int32 var_1335 = const()[name = string("op_1335"), val = int32(-1)]; fp16 const_21_promoted_to_fp16 = const()[name = string("const_21_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1337_cast_fp16 = mul(x = x_33_cast_fp16, y = const_21_promoted_to_fp16)[name = string("op_1337_cast_fp16")]; bool input_51_interleave_0 = const()[name = string("input_51_interleave_0"), val = bool(false)]; tensor input_51_cast_fp16 = concat(axis = var_1335, interleave = input_51_interleave_0, values = (x_33_cast_fp16, var_1337_cast_fp16))[name = string("input_51_cast_fp16")]; tensor normed_49_axes_0 = const()[name = string("normed_49_axes_0"), val = tensor([-1])]; fp16 var_1332_to_fp16 = const()[name = string("op_1332_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_49_cast_fp16 = layer_norm(axes = normed_49_axes_0, epsilon = var_1332_to_fp16, x = input_51_cast_fp16)[name = string("normed_49_cast_fp16")]; tensor var_1342_split_sizes_0 = const()[name = string("op_1342_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1342_axis_0 = const()[name = string("op_1342_axis_0"), val = int32(-1)]; tensor var_1342_cast_fp16_0, tensor var_1342_cast_fp16_1 = split(axis = var_1342_axis_0, split_sizes = var_1342_split_sizes_0, x = normed_49_cast_fp16)[name = string("op_1342_cast_fp16")]; tensor layers_1_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_1_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(142985600)))]; tensor var_1344_cast_fp16 = mul(x = var_1342_cast_fp16_0, y = layers_1_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("op_1344_cast_fp16")]; tensor gate_5 = linear(bias = linear_5_bias_0, weight = layers_1_mlp_gate_proj_weight_palettized, x = var_1344_cast_fp16)[name = string("linear_14")]; tensor up_3 = linear(bias = linear_5_bias_0, weight = layers_1_mlp_up_proj_weight_palettized, x = var_1344_cast_fp16)[name = string("linear_15")]; string gate_7_mode_0 = const()[name = string("gate_7_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_7 = gelu(mode = gate_7_mode_0, x = gate_5)[name = string("gate_7")]; tensor input_55 = mul(x = gate_7, y = up_3)[name = string("input_55")]; tensor x_35 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_1_mlp_down_proj_weight_palettized, x = input_55)[name = string("linear_16")]; int32 var_1366 = const()[name = string("op_1366"), val = int32(-1)]; fp16 const_22_promoted = const()[name = string("const_22_promoted"), val = fp16(-0x1p+0)]; tensor var_1368 = mul(x = x_35, y = const_22_promoted)[name = string("op_1368")]; bool input_57_interleave_0 = const()[name = string("input_57_interleave_0"), val = bool(false)]; tensor input_57 = concat(axis = var_1366, interleave = input_57_interleave_0, values = (x_35, var_1368))[name = string("input_57")]; tensor normed_53_axes_0 = const()[name = string("normed_53_axes_0"), val = tensor([-1])]; fp16 var_1363_to_fp16 = const()[name = string("op_1363_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_53_cast_fp16 = layer_norm(axes = normed_53_axes_0, epsilon = var_1363_to_fp16, x = input_57)[name = string("normed_53_cast_fp16")]; tensor var_1373_split_sizes_0 = const()[name = string("op_1373_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1373_axis_0 = const()[name = string("op_1373_axis_0"), val = int32(-1)]; tensor var_1373_0, tensor var_1373_1 = split(axis = var_1373_axis_0, split_sizes = var_1373_split_sizes_0, x = normed_53_cast_fp16)[name = string("op_1373")]; tensor hidden_states_11 = mul(x = var_1373_0, y = layers_1_post_feedforward_layernorm_weight)[name = string("hidden_states_11")]; tensor hidden_states_13_cast_fp16 = add(x = x_33_cast_fp16, y = hidden_states_11)[name = string("hidden_states_13_cast_fp16")]; tensor per_layer_slice_3_begin_0 = const()[name = string("per_layer_slice_3_begin_0"), val = tensor([0, 0, 256])]; tensor per_layer_slice_3_end_0 = const()[name = string("per_layer_slice_3_end_0"), val = tensor([1, 8, 512])]; tensor per_layer_slice_3_end_mask_0 = const()[name = string("per_layer_slice_3_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_3_cast_fp16 = slice_by_index(begin = per_layer_slice_3_begin_0, end = per_layer_slice_3_end_0, end_mask = per_layer_slice_3_end_mask_0, x = per_layer_combined_out)[name = string("per_layer_slice_3_cast_fp16")]; tensor gated_5 = linear(bias = linear_2_bias_0, weight = layers_1_per_layer_input_gate_weight_palettized, x = hidden_states_13_cast_fp16)[name = string("linear_17")]; string gated_7_mode_0 = const()[name = string("gated_7_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_7 = gelu(mode = gated_7_mode_0, x = gated_5)[name = string("gated_7")]; tensor input_61_cast_fp16 = mul(x = gated_7, y = per_layer_slice_3_cast_fp16)[name = string("input_61_cast_fp16")]; tensor layers_1_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(142988736))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(143185408))))[name = string("layers_1_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_18_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_1_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_61_cast_fp16)[name = string("linear_18_cast_fp16")]; int32 var_1411 = const()[name = string("op_1411"), val = int32(-1)]; fp16 const_23_promoted_to_fp16 = const()[name = string("const_23_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1413_cast_fp16 = mul(x = linear_18_cast_fp16, y = const_23_promoted_to_fp16)[name = string("op_1413_cast_fp16")]; bool input_63_interleave_0 = const()[name = string("input_63_interleave_0"), val = bool(false)]; tensor input_63_cast_fp16 = concat(axis = var_1411, interleave = input_63_interleave_0, values = (linear_18_cast_fp16, var_1413_cast_fp16))[name = string("input_63_cast_fp16")]; tensor normed_57_axes_0 = const()[name = string("normed_57_axes_0"), val = tensor([-1])]; fp16 var_1408_to_fp16 = const()[name = string("op_1408_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_57_cast_fp16 = layer_norm(axes = normed_57_axes_0, epsilon = var_1408_to_fp16, x = input_63_cast_fp16)[name = string("normed_57_cast_fp16")]; tensor var_1418_split_sizes_0 = const()[name = string("op_1418_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1418_axis_0 = const()[name = string("op_1418_axis_0"), val = int32(-1)]; tensor var_1418_cast_fp16_0, tensor var_1418_cast_fp16_1 = split(axis = var_1418_axis_0, split_sizes = var_1418_split_sizes_0, x = normed_57_cast_fp16)[name = string("op_1418_cast_fp16")]; tensor layers_1_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_1_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(143187008)))]; tensor hidden_states_15_cast_fp16 = mul(x = var_1418_cast_fp16_0, y = layers_1_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_15_cast_fp16")]; tensor hidden_states_17_cast_fp16 = add(x = hidden_states_13_cast_fp16, y = hidden_states_15_cast_fp16)[name = string("hidden_states_17_cast_fp16")]; tensor const_24_promoted_to_fp16 = const()[name = string("const_24_promoted_to_fp16"), val = tensor([0x1.c8p-3])]; tensor x_39_cast_fp16 = mul(x = hidden_states_17_cast_fp16, y = const_24_promoted_to_fp16)[name = string("x_39_cast_fp16")]; int32 var_1433 = const()[name = string("op_1433"), val = int32(-1)]; fp16 const_25_promoted_to_fp16 = const()[name = string("const_25_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1435_cast_fp16 = mul(x = x_39_cast_fp16, y = const_25_promoted_to_fp16)[name = string("op_1435_cast_fp16")]; bool input_65_interleave_0 = const()[name = string("input_65_interleave_0"), val = bool(false)]; tensor input_65_cast_fp16 = concat(axis = var_1433, interleave = input_65_interleave_0, values = (x_39_cast_fp16, var_1435_cast_fp16))[name = string("input_65_cast_fp16")]; tensor normed_61_axes_0 = const()[name = string("normed_61_axes_0"), val = tensor([-1])]; fp16 var_1430_to_fp16 = const()[name = string("op_1430_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_61_cast_fp16 = layer_norm(axes = normed_61_axes_0, epsilon = var_1430_to_fp16, x = input_65_cast_fp16)[name = string("normed_61_cast_fp16")]; tensor var_1440_split_sizes_0 = const()[name = string("op_1440_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1440_axis_0 = const()[name = string("op_1440_axis_0"), val = int32(-1)]; tensor var_1440_cast_fp16_0, tensor var_1440_cast_fp16_1 = split(axis = var_1440_axis_0, split_sizes = var_1440_split_sizes_0, x = normed_61_cast_fp16)[name = string("op_1440_cast_fp16")]; tensor layers_2_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_2_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(143190144)))]; tensor var_1442_cast_fp16 = mul(x = var_1440_cast_fp16_0, y = layers_2_input_layernorm_weight_promoted_to_fp16)[name = string("op_1442_cast_fp16")]; tensor var_1450 = linear(bias = linear_1_bias_0, weight = layers_2_self_attn_q_proj_weight_palettized, x = var_1442_cast_fp16)[name = string("linear_19")]; tensor var_1455 = const()[name = string("op_1455"), val = tensor([1, 8, 8, 256])]; tensor var_1456 = reshape(shape = var_1455, x = var_1450)[name = string("op_1456")]; tensor var_1461 = const()[name = string("op_1461"), val = tensor([0, 2, 1, 3])]; int32 var_1478 = const()[name = string("op_1478"), val = int32(-1)]; fp16 const_26_promoted = const()[name = string("const_26_promoted"), val = fp16(-0x1p+0)]; tensor var_1462 = transpose(perm = var_1461, x = var_1456)[name = string("transpose_59")]; tensor var_1480 = mul(x = var_1462, y = const_26_promoted)[name = string("op_1480")]; bool input_69_interleave_0 = const()[name = string("input_69_interleave_0"), val = bool(false)]; tensor input_69 = concat(axis = var_1478, interleave = input_69_interleave_0, values = (var_1462, var_1480))[name = string("input_69")]; tensor normed_65_axes_0 = const()[name = string("normed_65_axes_0"), val = tensor([-1])]; fp16 var_1475_to_fp16 = const()[name = string("op_1475_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_65_cast_fp16 = layer_norm(axes = normed_65_axes_0, epsilon = var_1475_to_fp16, x = input_69)[name = string("normed_65_cast_fp16")]; tensor var_1485_split_sizes_0 = const()[name = string("op_1485_split_sizes_0"), val = tensor([256, 256])]; int32 var_1485_axis_0 = const()[name = string("op_1485_axis_0"), val = int32(-1)]; tensor var_1485_0, tensor var_1485_1 = split(axis = var_1485_axis_0, split_sizes = var_1485_split_sizes_0, x = normed_65_cast_fp16)[name = string("op_1485")]; tensor q_19 = mul(x = var_1485_0, y = layers_2_self_attn_q_norm_weight)[name = string("q_19")]; tensor var_1488_cast_fp16 = mul(x = q_19, y = cos_s)[name = string("op_1488_cast_fp16")]; tensor var_1489_split_sizes_0 = const()[name = string("op_1489_split_sizes_0"), val = tensor([128, 128])]; int32 var_1489_axis_0 = const()[name = string("op_1489_axis_0"), val = int32(-1)]; tensor var_1489_0, tensor var_1489_1 = split(axis = var_1489_axis_0, split_sizes = var_1489_split_sizes_0, x = q_19)[name = string("op_1489")]; fp16 const_27_promoted = const()[name = string("const_27_promoted"), val = fp16(-0x1p+0)]; tensor var_1491 = mul(x = var_1489_1, y = const_27_promoted)[name = string("op_1491")]; int32 var_1493 = const()[name = string("op_1493"), val = int32(-1)]; bool var_1494_interleave_0 = const()[name = string("op_1494_interleave_0"), val = bool(false)]; tensor var_1494 = concat(axis = var_1493, interleave = var_1494_interleave_0, values = (var_1491, var_1489_0))[name = string("op_1494")]; tensor var_1495_cast_fp16 = mul(x = var_1494, y = sin_s)[name = string("op_1495_cast_fp16")]; tensor q_23_cast_fp16 = add(x = var_1488_cast_fp16, y = var_1495_cast_fp16)[name = string("q_23_cast_fp16")]; tensor var_1500 = linear(bias = linear_2_bias_0, weight = layers_2_self_attn_k_proj_weight_palettized, x = var_1442_cast_fp16)[name = string("linear_20")]; tensor var_1505 = const()[name = string("op_1505"), val = tensor([1, 8, 1, 256])]; tensor var_1506 = reshape(shape = var_1505, x = var_1500)[name = string("op_1506")]; tensor var_1511 = const()[name = string("op_1511"), val = tensor([0, 2, 1, 3])]; tensor var_1520 = linear(bias = linear_2_bias_0, weight = layers_2_self_attn_v_proj_weight_palettized, x = var_1442_cast_fp16)[name = string("linear_21")]; tensor var_1525 = const()[name = string("op_1525"), val = tensor([1, 8, 1, 256])]; tensor var_1526 = reshape(shape = var_1525, x = var_1520)[name = string("op_1526")]; tensor var_1531 = const()[name = string("op_1531"), val = tensor([0, 2, 1, 3])]; int32 var_1548 = const()[name = string("op_1548"), val = int32(-1)]; fp16 const_28_promoted = const()[name = string("const_28_promoted"), val = fp16(-0x1p+0)]; tensor var_1512 = transpose(perm = var_1511, x = var_1506)[name = string("transpose_58")]; tensor var_1550 = mul(x = var_1512, y = const_28_promoted)[name = string("op_1550")]; bool input_71_interleave_0 = const()[name = string("input_71_interleave_0"), val = bool(false)]; tensor input_71 = concat(axis = var_1548, interleave = input_71_interleave_0, values = (var_1512, var_1550))[name = string("input_71")]; tensor normed_69_axes_0 = const()[name = string("normed_69_axes_0"), val = tensor([-1])]; fp16 var_1545_to_fp16 = const()[name = string("op_1545_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_69_cast_fp16 = layer_norm(axes = normed_69_axes_0, epsilon = var_1545_to_fp16, x = input_71)[name = string("normed_69_cast_fp16")]; tensor var_1555_split_sizes_0 = const()[name = string("op_1555_split_sizes_0"), val = tensor([256, 256])]; int32 var_1555_axis_0 = const()[name = string("op_1555_axis_0"), val = int32(-1)]; tensor var_1555_0, tensor var_1555_1 = split(axis = var_1555_axis_0, split_sizes = var_1555_split_sizes_0, x = normed_69_cast_fp16)[name = string("op_1555")]; tensor q_21 = mul(x = var_1555_0, y = layers_2_self_attn_k_norm_weight)[name = string("q_21")]; fp16 var_1558_promoted = const()[name = string("op_1558_promoted"), val = fp16(0x1p+1)]; tensor var_1532 = transpose(perm = var_1531, x = var_1526)[name = string("transpose_57")]; tensor var_1559 = pow(x = var_1532, y = var_1558_promoted)[name = string("op_1559")]; tensor var_1564_axes_0 = const()[name = string("op_1564_axes_0"), val = tensor([-1])]; bool var_1564_keep_dims_0 = const()[name = string("op_1564_keep_dims_0"), val = bool(true)]; tensor var_1564 = reduce_mean(axes = var_1564_axes_0, keep_dims = var_1564_keep_dims_0, x = var_1559)[name = string("op_1564")]; fp16 var_1566_to_fp16 = const()[name = string("op_1566_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_5_cast_fp16 = add(x = var_1564, y = var_1566_to_fp16)[name = string("mean_sq_5_cast_fp16")]; fp32 var_1568_epsilon_0 = const()[name = string("op_1568_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_1568_cast_fp16 = rsqrt(epsilon = var_1568_epsilon_0, x = mean_sq_5_cast_fp16)[name = string("op_1568_cast_fp16")]; tensor input_75_cast_fp16 = mul(x = var_1532, y = var_1568_cast_fp16)[name = string("input_75_cast_fp16")]; tensor var_1570_cast_fp16 = mul(x = q_21, y = cos_s)[name = string("op_1570_cast_fp16")]; tensor var_1571_split_sizes_0 = const()[name = string("op_1571_split_sizes_0"), val = tensor([128, 128])]; int32 var_1571_axis_0 = const()[name = string("op_1571_axis_0"), val = int32(-1)]; tensor var_1571_0, tensor var_1571_1 = split(axis = var_1571_axis_0, split_sizes = var_1571_split_sizes_0, x = q_21)[name = string("op_1571")]; fp16 const_29_promoted = const()[name = string("const_29_promoted"), val = fp16(-0x1p+0)]; tensor var_1573 = mul(x = var_1571_1, y = const_29_promoted)[name = string("op_1573")]; int32 var_1575 = const()[name = string("op_1575"), val = int32(-1)]; bool var_1576_interleave_0 = const()[name = string("op_1576_interleave_0"), val = bool(false)]; tensor var_1576 = concat(axis = var_1575, interleave = var_1576_interleave_0, values = (var_1573, var_1571_0))[name = string("op_1576")]; tensor var_1577_cast_fp16 = mul(x = var_1576, y = sin_s)[name = string("op_1577_cast_fp16")]; tensor input_73_cast_fp16 = add(x = var_1570_cast_fp16, y = var_1577_cast_fp16)[name = string("input_73_cast_fp16")]; tensor k_padded_5_pad_0 = const()[name = string("k_padded_5_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_5_mode_0 = const()[name = string("k_padded_5_mode_0"), val = string("constant")]; fp16 const_30_to_fp16 = const()[name = string("const_30_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_5_cast_fp16 = pad(constant_val = const_30_to_fp16, mode = k_padded_5_mode_0, pad = k_padded_5_pad_0, x = input_73_cast_fp16)[name = string("k_padded_5_cast_fp16")]; tensor v_padded_5_pad_0 = const()[name = string("v_padded_5_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_5_mode_0 = const()[name = string("v_padded_5_mode_0"), val = string("constant")]; fp16 const_31_to_fp16 = const()[name = string("const_31_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_5_cast_fp16 = pad(constant_val = const_31_to_fp16, mode = v_padded_5_mode_0, pad = v_padded_5_pad_0, x = input_75_cast_fp16)[name = string("v_padded_5_cast_fp16")]; tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([4])]; tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([5])]; int32 concat_26_axis_0 = const()[name = string("concat_26_axis_0"), val = int32(0)]; bool concat_26_interleave_0 = const()[name = string("concat_26_interleave_0"), val = bool(false)]; tensor concat_26 = concat(axis = concat_26_axis_0, interleave = concat_26_interleave_0, values = (expand_dims_24, expand_dims_25, ring_pos, expand_dims_27))[name = string("concat_26")]; tensor concat_27_values1_0 = const()[name = string("concat_27_values1_0"), val = tensor([0])]; tensor concat_27_values3_0 = const()[name = string("concat_27_values3_0"), val = tensor([0])]; int32 concat_27_axis_0 = const()[name = string("concat_27_axis_0"), val = int32(0)]; bool concat_27_interleave_0 = const()[name = string("concat_27_interleave_0"), val = bool(false)]; tensor concat_27 = concat(axis = concat_27_axis_0, interleave = concat_27_interleave_0, values = (expand_dims_28, concat_27_values1_0, var_716, concat_27_values3_0))[name = string("concat_27")]; tensor kv_cache_sliding_internal_tensor_assign_5_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_5_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_5_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_26, begin_mask = kv_cache_sliding_internal_tensor_assign_5_begin_mask_0, end = concat_27, end_mask = kv_cache_sliding_internal_tensor_assign_5_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_5_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_5_stride_0, update = k_padded_5_cast_fp16, x = coreml_update_state_19)[name = string("kv_cache_sliding_internal_tensor_assign_5_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_5_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_20_write_state")]; tensor coreml_update_state_20 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_20")]; tensor expand_dims_30 = const()[name = string("expand_dims_30"), val = tensor([5])]; tensor expand_dims_31 = const()[name = string("expand_dims_31"), val = tensor([0])]; tensor expand_dims_33 = const()[name = string("expand_dims_33"), val = tensor([0])]; tensor expand_dims_34 = const()[name = string("expand_dims_34"), val = tensor([6])]; int32 concat_30_axis_0 = const()[name = string("concat_30_axis_0"), val = int32(0)]; bool concat_30_interleave_0 = const()[name = string("concat_30_interleave_0"), val = bool(false)]; tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (expand_dims_30, expand_dims_31, ring_pos, expand_dims_33))[name = string("concat_30")]; tensor concat_31_values1_0 = const()[name = string("concat_31_values1_0"), val = tensor([0])]; tensor concat_31_values3_0 = const()[name = string("concat_31_values3_0"), val = tensor([0])]; int32 concat_31_axis_0 = const()[name = string("concat_31_axis_0"), val = int32(0)]; bool concat_31_interleave_0 = const()[name = string("concat_31_interleave_0"), val = bool(false)]; tensor concat_31 = concat(axis = concat_31_axis_0, interleave = concat_31_interleave_0, values = (expand_dims_34, concat_31_values1_0, var_716, concat_31_values3_0))[name = string("concat_31")]; tensor kv_cache_sliding_internal_tensor_assign_6_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_6_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_6_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_30, begin_mask = kv_cache_sliding_internal_tensor_assign_6_begin_mask_0, end = concat_31, end_mask = kv_cache_sliding_internal_tensor_assign_6_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_6_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_6_stride_0, update = v_padded_5_cast_fp16, x = coreml_update_state_20)[name = string("kv_cache_sliding_internal_tensor_assign_6_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_6_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_21_write_state")]; tensor coreml_update_state_21 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_21")]; tensor var_1644_begin_0 = const()[name = string("op_1644_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_1644_end_0 = const()[name = string("op_1644_end_0"), val = tensor([5, 1, 512, 512])]; tensor var_1644_end_mask_0 = const()[name = string("op_1644_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1644_cast_fp16 = slice_by_index(begin = var_1644_begin_0, end = var_1644_end_0, end_mask = var_1644_end_mask_0, x = coreml_update_state_21)[name = string("op_1644_cast_fp16")]; tensor K_sliding_slice_5_begin_0 = const()[name = string("K_sliding_slice_5_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_sliding_slice_5_end_0 = const()[name = string("K_sliding_slice_5_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_sliding_slice_5_end_mask_0 = const()[name = string("K_sliding_slice_5_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_sliding_slice_5_cast_fp16 = slice_by_index(begin = K_sliding_slice_5_begin_0, end = K_sliding_slice_5_end_0, end_mask = K_sliding_slice_5_end_mask_0, x = var_1644_cast_fp16)[name = string("K_sliding_slice_5_cast_fp16")]; tensor var_1664_begin_0 = const()[name = string("op_1664_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_1664_end_0 = const()[name = string("op_1664_end_0"), val = tensor([6, 1, 512, 512])]; tensor var_1664_end_mask_0 = const()[name = string("op_1664_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1664_cast_fp16 = slice_by_index(begin = var_1664_begin_0, end = var_1664_end_0, end_mask = var_1664_end_mask_0, x = coreml_update_state_21)[name = string("op_1664_cast_fp16")]; tensor V_for_attn_5_begin_0 = const()[name = string("V_for_attn_5_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_5_end_0 = const()[name = string("V_for_attn_5_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_5_end_mask_0 = const()[name = string("V_for_attn_5_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_5_cast_fp16 = slice_by_index(begin = V_for_attn_5_begin_0, end = V_for_attn_5_end_0, end_mask = V_for_attn_5_end_mask_0, x = var_1664_cast_fp16)[name = string("V_for_attn_5_cast_fp16")]; tensor transpose_8_perm_0 = const()[name = string("transpose_8_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_4_reps_0 = const()[name = string("tile_4_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_8_cast_fp16 = transpose(perm = transpose_8_perm_0, x = K_sliding_slice_5_cast_fp16)[name = string("transpose_56")]; tensor tile_4_cast_fp16 = tile(reps = tile_4_reps_0, x = transpose_8_cast_fp16)[name = string("tile_4_cast_fp16")]; tensor concat_32 = const()[name = string("concat_32"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_8_cast_fp16 = reshape(shape = concat_32, x = tile_4_cast_fp16)[name = string("reshape_8_cast_fp16")]; tensor transpose_9_perm_0 = const()[name = string("transpose_9_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_33 = const()[name = string("concat_33"), val = tensor([-1, 1, 512, 256])]; tensor transpose_9_cast_fp16 = transpose(perm = transpose_9_perm_0, x = reshape_8_cast_fp16)[name = string("transpose_55")]; tensor reshape_9_cast_fp16 = reshape(shape = concat_33, x = transpose_9_cast_fp16)[name = string("reshape_9_cast_fp16")]; tensor transpose_34_perm_0 = const()[name = string("transpose_34_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_10_perm_0 = const()[name = string("transpose_10_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_5_reps_0 = const()[name = string("tile_5_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_10_cast_fp16 = transpose(perm = transpose_10_perm_0, x = V_for_attn_5_cast_fp16)[name = string("transpose_54")]; tensor tile_5_cast_fp16 = tile(reps = tile_5_reps_0, x = transpose_10_cast_fp16)[name = string("tile_5_cast_fp16")]; tensor concat_34 = const()[name = string("concat_34"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_10_cast_fp16 = reshape(shape = concat_34, x = tile_5_cast_fp16)[name = string("reshape_10_cast_fp16")]; tensor transpose_11_perm_0 = const()[name = string("transpose_11_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_35 = const()[name = string("concat_35"), val = tensor([-1, 1, 512, 256])]; tensor transpose_11_cast_fp16 = transpose(perm = transpose_11_perm_0, x = reshape_10_cast_fp16)[name = string("transpose_53")]; tensor reshape_11_cast_fp16 = reshape(shape = concat_35, x = transpose_11_cast_fp16)[name = string("reshape_11_cast_fp16")]; tensor V_expanded_5_perm_0 = const()[name = string("V_expanded_5_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_9_transpose_x_0 = const()[name = string("attn_weights_9_transpose_x_0"), val = bool(false)]; bool attn_weights_9_transpose_y_0 = const()[name = string("attn_weights_9_transpose_y_0"), val = bool(false)]; tensor transpose_34_cast_fp16 = transpose(perm = transpose_34_perm_0, x = reshape_9_cast_fp16)[name = string("transpose_52")]; tensor attn_weights_9_cast_fp16 = matmul(transpose_x = attn_weights_9_transpose_x_0, transpose_y = attn_weights_9_transpose_y_0, x = q_23_cast_fp16, y = transpose_34_cast_fp16)[name = string("attn_weights_9_cast_fp16")]; tensor x_47_cast_fp16 = add(x = attn_weights_9_cast_fp16, y = causal_mask_sliding)[name = string("x_47_cast_fp16")]; tensor reduce_max_2_axes_0 = const()[name = string("reduce_max_2_axes_0"), val = tensor([-1])]; bool reduce_max_2_keep_dims_0 = const()[name = string("reduce_max_2_keep_dims_0"), val = bool(true)]; tensor reduce_max_2 = reduce_max(axes = reduce_max_2_axes_0, keep_dims = reduce_max_2_keep_dims_0, x = x_47_cast_fp16)[name = string("reduce_max_2")]; tensor var_1709 = sub(x = x_47_cast_fp16, y = reduce_max_2)[name = string("op_1709")]; tensor var_1715 = exp(x = var_1709)[name = string("op_1715")]; tensor var_1725_axes_0 = const()[name = string("op_1725_axes_0"), val = tensor([-1])]; bool var_1725_keep_dims_0 = const()[name = string("op_1725_keep_dims_0"), val = bool(true)]; tensor var_1725 = reduce_sum(axes = var_1725_axes_0, keep_dims = var_1725_keep_dims_0, x = var_1715)[name = string("op_1725")]; tensor var_1731_cast_fp16 = real_div(x = var_1715, y = var_1725)[name = string("op_1731_cast_fp16")]; bool attn_output_9_transpose_x_0 = const()[name = string("attn_output_9_transpose_x_0"), val = bool(false)]; bool attn_output_9_transpose_y_0 = const()[name = string("attn_output_9_transpose_y_0"), val = bool(false)]; tensor V_expanded_5_cast_fp16 = transpose(perm = V_expanded_5_perm_0, x = reshape_11_cast_fp16)[name = string("transpose_51")]; tensor attn_output_9_cast_fp16 = matmul(transpose_x = attn_output_9_transpose_x_0, transpose_y = attn_output_9_transpose_y_0, x = var_1731_cast_fp16, y = V_expanded_5_cast_fp16)[name = string("attn_output_9_cast_fp16")]; tensor var_1742 = const()[name = string("op_1742"), val = tensor([0, 2, 1, 3])]; tensor var_1749 = const()[name = string("op_1749"), val = tensor([1, 8, -1])]; tensor var_1743_cast_fp16 = transpose(perm = var_1742, x = attn_output_9_cast_fp16)[name = string("transpose_50")]; tensor input_77_cast_fp16 = reshape(shape = var_1749, x = var_1743_cast_fp16)[name = string("input_77_cast_fp16")]; tensor layers_2_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(143193280))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144766208))))[name = string("layers_2_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_22_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_2_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_77_cast_fp16)[name = string("linear_22_cast_fp16")]; int32 var_1758 = const()[name = string("op_1758"), val = int32(-1)]; fp16 const_32_promoted_to_fp16 = const()[name = string("const_32_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1760_cast_fp16 = mul(x = linear_22_cast_fp16, y = const_32_promoted_to_fp16)[name = string("op_1760_cast_fp16")]; bool input_79_interleave_0 = const()[name = string("input_79_interleave_0"), val = bool(false)]; tensor input_79_cast_fp16 = concat(axis = var_1758, interleave = input_79_interleave_0, values = (linear_22_cast_fp16, var_1760_cast_fp16))[name = string("input_79_cast_fp16")]; tensor normed_73_axes_0 = const()[name = string("normed_73_axes_0"), val = tensor([-1])]; fp16 var_1755_to_fp16 = const()[name = string("op_1755_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_73_cast_fp16 = layer_norm(axes = normed_73_axes_0, epsilon = var_1755_to_fp16, x = input_79_cast_fp16)[name = string("normed_73_cast_fp16")]; tensor var_1765_split_sizes_0 = const()[name = string("op_1765_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1765_axis_0 = const()[name = string("op_1765_axis_0"), val = int32(-1)]; tensor var_1765_cast_fp16_0, tensor var_1765_cast_fp16_1 = split(axis = var_1765_axis_0, split_sizes = var_1765_split_sizes_0, x = normed_73_cast_fp16)[name = string("op_1765_cast_fp16")]; tensor layers_2_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_2_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144767808)))]; tensor attn_output_11_cast_fp16 = mul(x = var_1765_cast_fp16_0, y = layers_2_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_11_cast_fp16")]; tensor x_53_cast_fp16 = add(x = x_39_cast_fp16, y = attn_output_11_cast_fp16)[name = string("x_53_cast_fp16")]; int32 var_1774 = const()[name = string("op_1774"), val = int32(-1)]; fp16 const_33_promoted_to_fp16 = const()[name = string("const_33_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1776_cast_fp16 = mul(x = x_53_cast_fp16, y = const_33_promoted_to_fp16)[name = string("op_1776_cast_fp16")]; bool input_81_interleave_0 = const()[name = string("input_81_interleave_0"), val = bool(false)]; tensor input_81_cast_fp16 = concat(axis = var_1774, interleave = input_81_interleave_0, values = (x_53_cast_fp16, var_1776_cast_fp16))[name = string("input_81_cast_fp16")]; tensor normed_77_axes_0 = const()[name = string("normed_77_axes_0"), val = tensor([-1])]; fp16 var_1771_to_fp16 = const()[name = string("op_1771_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_77_cast_fp16 = layer_norm(axes = normed_77_axes_0, epsilon = var_1771_to_fp16, x = input_81_cast_fp16)[name = string("normed_77_cast_fp16")]; tensor var_1781_split_sizes_0 = const()[name = string("op_1781_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1781_axis_0 = const()[name = string("op_1781_axis_0"), val = int32(-1)]; tensor var_1781_cast_fp16_0, tensor var_1781_cast_fp16_1 = split(axis = var_1781_axis_0, split_sizes = var_1781_split_sizes_0, x = normed_77_cast_fp16)[name = string("op_1781_cast_fp16")]; tensor layers_2_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_2_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144770944)))]; tensor var_1783_cast_fp16 = mul(x = var_1781_cast_fp16_0, y = layers_2_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("op_1783_cast_fp16")]; tensor gate_9 = linear(bias = linear_5_bias_0, weight = layers_2_mlp_gate_proj_weight_palettized, x = var_1783_cast_fp16)[name = string("linear_23")]; tensor up_5 = linear(bias = linear_5_bias_0, weight = layers_2_mlp_up_proj_weight_palettized, x = var_1783_cast_fp16)[name = string("linear_24")]; string gate_11_mode_0 = const()[name = string("gate_11_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_11 = gelu(mode = gate_11_mode_0, x = gate_9)[name = string("gate_11")]; tensor input_85 = mul(x = gate_11, y = up_5)[name = string("input_85")]; tensor x_55 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_2_mlp_down_proj_weight_palettized, x = input_85)[name = string("linear_25")]; int32 var_1805 = const()[name = string("op_1805"), val = int32(-1)]; fp16 const_34_promoted = const()[name = string("const_34_promoted"), val = fp16(-0x1p+0)]; tensor var_1807 = mul(x = x_55, y = const_34_promoted)[name = string("op_1807")]; bool input_87_interleave_0 = const()[name = string("input_87_interleave_0"), val = bool(false)]; tensor input_87 = concat(axis = var_1805, interleave = input_87_interleave_0, values = (x_55, var_1807))[name = string("input_87")]; tensor normed_81_axes_0 = const()[name = string("normed_81_axes_0"), val = tensor([-1])]; fp16 var_1802_to_fp16 = const()[name = string("op_1802_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_81_cast_fp16 = layer_norm(axes = normed_81_axes_0, epsilon = var_1802_to_fp16, x = input_87)[name = string("normed_81_cast_fp16")]; tensor var_1812_split_sizes_0 = const()[name = string("op_1812_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1812_axis_0 = const()[name = string("op_1812_axis_0"), val = int32(-1)]; tensor var_1812_0, tensor var_1812_1 = split(axis = var_1812_axis_0, split_sizes = var_1812_split_sizes_0, x = normed_81_cast_fp16)[name = string("op_1812")]; tensor hidden_states_19 = mul(x = var_1812_0, y = layers_2_post_feedforward_layernorm_weight)[name = string("hidden_states_19")]; tensor hidden_states_21_cast_fp16 = add(x = x_53_cast_fp16, y = hidden_states_19)[name = string("hidden_states_21_cast_fp16")]; tensor per_layer_slice_5_begin_0 = const()[name = string("per_layer_slice_5_begin_0"), val = tensor([0, 0, 512])]; tensor per_layer_slice_5_end_0 = const()[name = string("per_layer_slice_5_end_0"), val = tensor([1, 8, 768])]; tensor per_layer_slice_5_end_mask_0 = const()[name = string("per_layer_slice_5_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_5_cast_fp16 = slice_by_index(begin = per_layer_slice_5_begin_0, end = per_layer_slice_5_end_0, end_mask = per_layer_slice_5_end_mask_0, x = per_layer_combined_out)[name = string("per_layer_slice_5_cast_fp16")]; tensor gated_9 = linear(bias = linear_2_bias_0, weight = layers_2_per_layer_input_gate_weight_palettized, x = hidden_states_21_cast_fp16)[name = string("linear_26")]; string gated_11_mode_0 = const()[name = string("gated_11_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_11 = gelu(mode = gated_11_mode_0, x = gated_9)[name = string("gated_11")]; tensor input_91_cast_fp16 = mul(x = gated_11, y = per_layer_slice_5_cast_fp16)[name = string("input_91_cast_fp16")]; tensor layers_2_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144774080))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144970752))))[name = string("layers_2_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_27_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_2_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_91_cast_fp16)[name = string("linear_27_cast_fp16")]; int32 var_1850 = const()[name = string("op_1850"), val = int32(-1)]; fp16 const_35_promoted_to_fp16 = const()[name = string("const_35_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1852_cast_fp16 = mul(x = linear_27_cast_fp16, y = const_35_promoted_to_fp16)[name = string("op_1852_cast_fp16")]; bool input_93_interleave_0 = const()[name = string("input_93_interleave_0"), val = bool(false)]; tensor input_93_cast_fp16 = concat(axis = var_1850, interleave = input_93_interleave_0, values = (linear_27_cast_fp16, var_1852_cast_fp16))[name = string("input_93_cast_fp16")]; tensor normed_85_axes_0 = const()[name = string("normed_85_axes_0"), val = tensor([-1])]; fp16 var_1847_to_fp16 = const()[name = string("op_1847_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_85_cast_fp16 = layer_norm(axes = normed_85_axes_0, epsilon = var_1847_to_fp16, x = input_93_cast_fp16)[name = string("normed_85_cast_fp16")]; tensor var_1857_split_sizes_0 = const()[name = string("op_1857_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1857_axis_0 = const()[name = string("op_1857_axis_0"), val = int32(-1)]; tensor var_1857_cast_fp16_0, tensor var_1857_cast_fp16_1 = split(axis = var_1857_axis_0, split_sizes = var_1857_split_sizes_0, x = normed_85_cast_fp16)[name = string("op_1857_cast_fp16")]; tensor layers_2_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_2_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144972352)))]; tensor hidden_states_23_cast_fp16 = mul(x = var_1857_cast_fp16_0, y = layers_2_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_23_cast_fp16")]; tensor hidden_states_25_cast_fp16 = add(x = hidden_states_21_cast_fp16, y = hidden_states_23_cast_fp16)[name = string("hidden_states_25_cast_fp16")]; tensor const_36_promoted_to_fp16 = const()[name = string("const_36_promoted_to_fp16"), val = tensor([0x1.96p-1])]; tensor x_59_cast_fp16 = mul(x = hidden_states_25_cast_fp16, y = const_36_promoted_to_fp16)[name = string("x_59_cast_fp16")]; int32 var_1872 = const()[name = string("op_1872"), val = int32(-1)]; fp16 const_37_promoted_to_fp16 = const()[name = string("const_37_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1874_cast_fp16 = mul(x = x_59_cast_fp16, y = const_37_promoted_to_fp16)[name = string("op_1874_cast_fp16")]; bool input_95_interleave_0 = const()[name = string("input_95_interleave_0"), val = bool(false)]; tensor input_95_cast_fp16 = concat(axis = var_1872, interleave = input_95_interleave_0, values = (x_59_cast_fp16, var_1874_cast_fp16))[name = string("input_95_cast_fp16")]; tensor normed_89_axes_0 = const()[name = string("normed_89_axes_0"), val = tensor([-1])]; fp16 var_1869_to_fp16 = const()[name = string("op_1869_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_89_cast_fp16 = layer_norm(axes = normed_89_axes_0, epsilon = var_1869_to_fp16, x = input_95_cast_fp16)[name = string("normed_89_cast_fp16")]; tensor var_1879_split_sizes_0 = const()[name = string("op_1879_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_1879_axis_0 = const()[name = string("op_1879_axis_0"), val = int32(-1)]; tensor var_1879_cast_fp16_0, tensor var_1879_cast_fp16_1 = split(axis = var_1879_axis_0, split_sizes = var_1879_split_sizes_0, x = normed_89_cast_fp16)[name = string("op_1879_cast_fp16")]; tensor layers_3_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_3_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144975488)))]; tensor var_1881_cast_fp16 = mul(x = var_1879_cast_fp16_0, y = layers_3_input_layernorm_weight_promoted_to_fp16)[name = string("op_1881_cast_fp16")]; tensor var_1889 = linear(bias = linear_1_bias_0, weight = layers_3_self_attn_q_proj_weight_palettized, x = var_1881_cast_fp16)[name = string("linear_28")]; tensor var_1894 = const()[name = string("op_1894"), val = tensor([1, 8, 8, 256])]; tensor var_1895 = reshape(shape = var_1894, x = var_1889)[name = string("op_1895")]; tensor var_1900 = const()[name = string("op_1900"), val = tensor([0, 2, 1, 3])]; int32 var_1917 = const()[name = string("op_1917"), val = int32(-1)]; fp16 const_38_promoted = const()[name = string("const_38_promoted"), val = fp16(-0x1p+0)]; tensor var_1901 = transpose(perm = var_1900, x = var_1895)[name = string("transpose_49")]; tensor var_1919 = mul(x = var_1901, y = const_38_promoted)[name = string("op_1919")]; bool input_99_interleave_0 = const()[name = string("input_99_interleave_0"), val = bool(false)]; tensor input_99 = concat(axis = var_1917, interleave = input_99_interleave_0, values = (var_1901, var_1919))[name = string("input_99")]; tensor normed_93_axes_0 = const()[name = string("normed_93_axes_0"), val = tensor([-1])]; fp16 var_1914_to_fp16 = const()[name = string("op_1914_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_93_cast_fp16 = layer_norm(axes = normed_93_axes_0, epsilon = var_1914_to_fp16, x = input_99)[name = string("normed_93_cast_fp16")]; tensor var_1924_split_sizes_0 = const()[name = string("op_1924_split_sizes_0"), val = tensor([256, 256])]; int32 var_1924_axis_0 = const()[name = string("op_1924_axis_0"), val = int32(-1)]; tensor var_1924_0, tensor var_1924_1 = split(axis = var_1924_axis_0, split_sizes = var_1924_split_sizes_0, x = normed_93_cast_fp16)[name = string("op_1924")]; tensor q_27 = mul(x = var_1924_0, y = layers_3_self_attn_q_norm_weight)[name = string("q_27")]; tensor var_1927_cast_fp16 = mul(x = q_27, y = cos_s)[name = string("op_1927_cast_fp16")]; tensor var_1928_split_sizes_0 = const()[name = string("op_1928_split_sizes_0"), val = tensor([128, 128])]; int32 var_1928_axis_0 = const()[name = string("op_1928_axis_0"), val = int32(-1)]; tensor var_1928_0, tensor var_1928_1 = split(axis = var_1928_axis_0, split_sizes = var_1928_split_sizes_0, x = q_27)[name = string("op_1928")]; fp16 const_39_promoted = const()[name = string("const_39_promoted"), val = fp16(-0x1p+0)]; tensor var_1930 = mul(x = var_1928_1, y = const_39_promoted)[name = string("op_1930")]; int32 var_1932 = const()[name = string("op_1932"), val = int32(-1)]; bool var_1933_interleave_0 = const()[name = string("op_1933_interleave_0"), val = bool(false)]; tensor var_1933 = concat(axis = var_1932, interleave = var_1933_interleave_0, values = (var_1930, var_1928_0))[name = string("op_1933")]; tensor var_1934_cast_fp16 = mul(x = var_1933, y = sin_s)[name = string("op_1934_cast_fp16")]; tensor q_31_cast_fp16 = add(x = var_1927_cast_fp16, y = var_1934_cast_fp16)[name = string("q_31_cast_fp16")]; tensor var_1939 = linear(bias = linear_2_bias_0, weight = layers_3_self_attn_k_proj_weight_palettized, x = var_1881_cast_fp16)[name = string("linear_29")]; tensor var_1944 = const()[name = string("op_1944"), val = tensor([1, 8, 1, 256])]; tensor var_1945 = reshape(shape = var_1944, x = var_1939)[name = string("op_1945")]; tensor var_1950 = const()[name = string("op_1950"), val = tensor([0, 2, 1, 3])]; tensor var_1959 = linear(bias = linear_2_bias_0, weight = layers_3_self_attn_v_proj_weight_palettized, x = var_1881_cast_fp16)[name = string("linear_30")]; tensor var_1964 = const()[name = string("op_1964"), val = tensor([1, 8, 1, 256])]; tensor var_1965 = reshape(shape = var_1964, x = var_1959)[name = string("op_1965")]; tensor var_1970 = const()[name = string("op_1970"), val = tensor([0, 2, 1, 3])]; int32 var_1987 = const()[name = string("op_1987"), val = int32(-1)]; fp16 const_40_promoted = const()[name = string("const_40_promoted"), val = fp16(-0x1p+0)]; tensor var_1951 = transpose(perm = var_1950, x = var_1945)[name = string("transpose_48")]; tensor var_1989 = mul(x = var_1951, y = const_40_promoted)[name = string("op_1989")]; bool input_101_interleave_0 = const()[name = string("input_101_interleave_0"), val = bool(false)]; tensor input_101 = concat(axis = var_1987, interleave = input_101_interleave_0, values = (var_1951, var_1989))[name = string("input_101")]; tensor normed_97_axes_0 = const()[name = string("normed_97_axes_0"), val = tensor([-1])]; fp16 var_1984_to_fp16 = const()[name = string("op_1984_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_97_cast_fp16 = layer_norm(axes = normed_97_axes_0, epsilon = var_1984_to_fp16, x = input_101)[name = string("normed_97_cast_fp16")]; tensor var_1994_split_sizes_0 = const()[name = string("op_1994_split_sizes_0"), val = tensor([256, 256])]; int32 var_1994_axis_0 = const()[name = string("op_1994_axis_0"), val = int32(-1)]; tensor var_1994_0, tensor var_1994_1 = split(axis = var_1994_axis_0, split_sizes = var_1994_split_sizes_0, x = normed_97_cast_fp16)[name = string("op_1994")]; tensor q_29 = mul(x = var_1994_0, y = layers_3_self_attn_k_norm_weight)[name = string("q_29")]; fp16 var_1997_promoted = const()[name = string("op_1997_promoted"), val = fp16(0x1p+1)]; tensor var_1971 = transpose(perm = var_1970, x = var_1965)[name = string("transpose_47")]; tensor var_1998 = pow(x = var_1971, y = var_1997_promoted)[name = string("op_1998")]; tensor var_2003_axes_0 = const()[name = string("op_2003_axes_0"), val = tensor([-1])]; bool var_2003_keep_dims_0 = const()[name = string("op_2003_keep_dims_0"), val = bool(true)]; tensor var_2003 = reduce_mean(axes = var_2003_axes_0, keep_dims = var_2003_keep_dims_0, x = var_1998)[name = string("op_2003")]; fp16 var_2005_to_fp16 = const()[name = string("op_2005_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_7_cast_fp16 = add(x = var_2003, y = var_2005_to_fp16)[name = string("mean_sq_7_cast_fp16")]; fp32 var_2007_epsilon_0 = const()[name = string("op_2007_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_2007_cast_fp16 = rsqrt(epsilon = var_2007_epsilon_0, x = mean_sq_7_cast_fp16)[name = string("op_2007_cast_fp16")]; tensor input_105_cast_fp16 = mul(x = var_1971, y = var_2007_cast_fp16)[name = string("input_105_cast_fp16")]; tensor var_2009_cast_fp16 = mul(x = q_29, y = cos_s)[name = string("op_2009_cast_fp16")]; tensor var_2010_split_sizes_0 = const()[name = string("op_2010_split_sizes_0"), val = tensor([128, 128])]; int32 var_2010_axis_0 = const()[name = string("op_2010_axis_0"), val = int32(-1)]; tensor var_2010_0, tensor var_2010_1 = split(axis = var_2010_axis_0, split_sizes = var_2010_split_sizes_0, x = q_29)[name = string("op_2010")]; fp16 const_41_promoted = const()[name = string("const_41_promoted"), val = fp16(-0x1p+0)]; tensor var_2012 = mul(x = var_2010_1, y = const_41_promoted)[name = string("op_2012")]; int32 var_2014 = const()[name = string("op_2014"), val = int32(-1)]; bool var_2015_interleave_0 = const()[name = string("op_2015_interleave_0"), val = bool(false)]; tensor var_2015 = concat(axis = var_2014, interleave = var_2015_interleave_0, values = (var_2012, var_2010_0))[name = string("op_2015")]; tensor var_2016_cast_fp16 = mul(x = var_2015, y = sin_s)[name = string("op_2016_cast_fp16")]; tensor input_103_cast_fp16 = add(x = var_2009_cast_fp16, y = var_2016_cast_fp16)[name = string("input_103_cast_fp16")]; tensor k_padded_7_pad_0 = const()[name = string("k_padded_7_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_7_mode_0 = const()[name = string("k_padded_7_mode_0"), val = string("constant")]; fp16 const_42_to_fp16 = const()[name = string("const_42_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_7_cast_fp16 = pad(constant_val = const_42_to_fp16, mode = k_padded_7_mode_0, pad = k_padded_7_pad_0, x = input_103_cast_fp16)[name = string("k_padded_7_cast_fp16")]; tensor v_padded_7_pad_0 = const()[name = string("v_padded_7_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_7_mode_0 = const()[name = string("v_padded_7_mode_0"), val = string("constant")]; fp16 const_43_to_fp16 = const()[name = string("const_43_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_7_cast_fp16 = pad(constant_val = const_43_to_fp16, mode = v_padded_7_mode_0, pad = v_padded_7_pad_0, x = input_105_cast_fp16)[name = string("v_padded_7_cast_fp16")]; tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([6])]; tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([7])]; int32 concat_38_axis_0 = const()[name = string("concat_38_axis_0"), val = int32(0)]; bool concat_38_interleave_0 = const()[name = string("concat_38_interleave_0"), val = bool(false)]; tensor concat_38 = concat(axis = concat_38_axis_0, interleave = concat_38_interleave_0, values = (expand_dims_36, expand_dims_37, ring_pos, expand_dims_39))[name = string("concat_38")]; tensor concat_39_values1_0 = const()[name = string("concat_39_values1_0"), val = tensor([0])]; tensor concat_39_values3_0 = const()[name = string("concat_39_values3_0"), val = tensor([0])]; int32 concat_39_axis_0 = const()[name = string("concat_39_axis_0"), val = int32(0)]; bool concat_39_interleave_0 = const()[name = string("concat_39_interleave_0"), val = bool(false)]; tensor concat_39 = concat(axis = concat_39_axis_0, interleave = concat_39_interleave_0, values = (expand_dims_40, concat_39_values1_0, var_716, concat_39_values3_0))[name = string("concat_39")]; tensor kv_cache_sliding_internal_tensor_assign_7_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_7_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_7_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_38, begin_mask = kv_cache_sliding_internal_tensor_assign_7_begin_mask_0, end = concat_39, end_mask = kv_cache_sliding_internal_tensor_assign_7_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_7_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_7_stride_0, update = k_padded_7_cast_fp16, x = coreml_update_state_21)[name = string("kv_cache_sliding_internal_tensor_assign_7_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_7_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_22_write_state")]; tensor coreml_update_state_22 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_22")]; tensor expand_dims_42 = const()[name = string("expand_dims_42"), val = tensor([7])]; tensor expand_dims_43 = const()[name = string("expand_dims_43"), val = tensor([0])]; tensor expand_dims_45 = const()[name = string("expand_dims_45"), val = tensor([0])]; tensor expand_dims_46 = const()[name = string("expand_dims_46"), val = tensor([8])]; int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_42, expand_dims_43, ring_pos, expand_dims_45))[name = string("concat_42")]; tensor concat_43_values1_0 = const()[name = string("concat_43_values1_0"), val = tensor([0])]; tensor concat_43_values3_0 = const()[name = string("concat_43_values3_0"), val = tensor([0])]; int32 concat_43_axis_0 = const()[name = string("concat_43_axis_0"), val = int32(0)]; bool concat_43_interleave_0 = const()[name = string("concat_43_interleave_0"), val = bool(false)]; tensor concat_43 = concat(axis = concat_43_axis_0, interleave = concat_43_interleave_0, values = (expand_dims_46, concat_43_values1_0, var_716, concat_43_values3_0))[name = string("concat_43")]; tensor kv_cache_sliding_internal_tensor_assign_8_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_8_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_8_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_42, begin_mask = kv_cache_sliding_internal_tensor_assign_8_begin_mask_0, end = concat_43, end_mask = kv_cache_sliding_internal_tensor_assign_8_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_8_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_8_stride_0, update = v_padded_7_cast_fp16, x = coreml_update_state_22)[name = string("kv_cache_sliding_internal_tensor_assign_8_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_8_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_23_write_state")]; tensor coreml_update_state_23 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_23")]; tensor var_2083_begin_0 = const()[name = string("op_2083_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_2083_end_0 = const()[name = string("op_2083_end_0"), val = tensor([7, 1, 512, 512])]; tensor var_2083_end_mask_0 = const()[name = string("op_2083_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2083_cast_fp16 = slice_by_index(begin = var_2083_begin_0, end = var_2083_end_0, end_mask = var_2083_end_mask_0, x = coreml_update_state_23)[name = string("op_2083_cast_fp16")]; tensor K_sliding_slice_7_begin_0 = const()[name = string("K_sliding_slice_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_sliding_slice_7_end_0 = const()[name = string("K_sliding_slice_7_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_sliding_slice_7_end_mask_0 = const()[name = string("K_sliding_slice_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_sliding_slice_7_cast_fp16 = slice_by_index(begin = K_sliding_slice_7_begin_0, end = K_sliding_slice_7_end_0, end_mask = K_sliding_slice_7_end_mask_0, x = var_2083_cast_fp16)[name = string("K_sliding_slice_7_cast_fp16")]; tensor var_2103_begin_0 = const()[name = string("op_2103_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_2103_end_0 = const()[name = string("op_2103_end_0"), val = tensor([8, 1, 512, 512])]; tensor var_2103_end_mask_0 = const()[name = string("op_2103_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2103_cast_fp16 = slice_by_index(begin = var_2103_begin_0, end = var_2103_end_0, end_mask = var_2103_end_mask_0, x = coreml_update_state_23)[name = string("op_2103_cast_fp16")]; tensor V_for_attn_7_begin_0 = const()[name = string("V_for_attn_7_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_7_end_0 = const()[name = string("V_for_attn_7_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_7_end_mask_0 = const()[name = string("V_for_attn_7_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_7_cast_fp16 = slice_by_index(begin = V_for_attn_7_begin_0, end = V_for_attn_7_end_0, end_mask = V_for_attn_7_end_mask_0, x = var_2103_cast_fp16)[name = string("V_for_attn_7_cast_fp16")]; tensor transpose_12_perm_0 = const()[name = string("transpose_12_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_6_reps_0 = const()[name = string("tile_6_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_12_cast_fp16 = transpose(perm = transpose_12_perm_0, x = K_sliding_slice_7_cast_fp16)[name = string("transpose_46")]; tensor tile_6_cast_fp16 = tile(reps = tile_6_reps_0, x = transpose_12_cast_fp16)[name = string("tile_6_cast_fp16")]; tensor concat_44 = const()[name = string("concat_44"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_12_cast_fp16 = reshape(shape = concat_44, x = tile_6_cast_fp16)[name = string("reshape_12_cast_fp16")]; tensor transpose_13_perm_0 = const()[name = string("transpose_13_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_45 = const()[name = string("concat_45"), val = tensor([-1, 1, 512, 256])]; tensor transpose_13_cast_fp16 = transpose(perm = transpose_13_perm_0, x = reshape_12_cast_fp16)[name = string("transpose_45")]; tensor reshape_13_cast_fp16 = reshape(shape = concat_45, x = transpose_13_cast_fp16)[name = string("reshape_13_cast_fp16")]; tensor transpose_35_perm_0 = const()[name = string("transpose_35_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_14_perm_0 = const()[name = string("transpose_14_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_7_reps_0 = const()[name = string("tile_7_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_14_cast_fp16 = transpose(perm = transpose_14_perm_0, x = V_for_attn_7_cast_fp16)[name = string("transpose_44")]; tensor tile_7_cast_fp16 = tile(reps = tile_7_reps_0, x = transpose_14_cast_fp16)[name = string("tile_7_cast_fp16")]; tensor concat_46 = const()[name = string("concat_46"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_14_cast_fp16 = reshape(shape = concat_46, x = tile_7_cast_fp16)[name = string("reshape_14_cast_fp16")]; tensor transpose_15_perm_0 = const()[name = string("transpose_15_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_47 = const()[name = string("concat_47"), val = tensor([-1, 1, 512, 256])]; tensor transpose_15_cast_fp16 = transpose(perm = transpose_15_perm_0, x = reshape_14_cast_fp16)[name = string("transpose_43")]; tensor reshape_15_cast_fp16 = reshape(shape = concat_47, x = transpose_15_cast_fp16)[name = string("reshape_15_cast_fp16")]; tensor V_expanded_7_perm_0 = const()[name = string("V_expanded_7_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_13_transpose_x_0 = const()[name = string("attn_weights_13_transpose_x_0"), val = bool(false)]; bool attn_weights_13_transpose_y_0 = const()[name = string("attn_weights_13_transpose_y_0"), val = bool(false)]; tensor transpose_35_cast_fp16 = transpose(perm = transpose_35_perm_0, x = reshape_13_cast_fp16)[name = string("transpose_42")]; tensor attn_weights_13_cast_fp16 = matmul(transpose_x = attn_weights_13_transpose_x_0, transpose_y = attn_weights_13_transpose_y_0, x = q_31_cast_fp16, y = transpose_35_cast_fp16)[name = string("attn_weights_13_cast_fp16")]; tensor x_67_cast_fp16 = add(x = attn_weights_13_cast_fp16, y = causal_mask_sliding)[name = string("x_67_cast_fp16")]; tensor reduce_max_3_axes_0 = const()[name = string("reduce_max_3_axes_0"), val = tensor([-1])]; bool reduce_max_3_keep_dims_0 = const()[name = string("reduce_max_3_keep_dims_0"), val = bool(true)]; tensor reduce_max_3 = reduce_max(axes = reduce_max_3_axes_0, keep_dims = reduce_max_3_keep_dims_0, x = x_67_cast_fp16)[name = string("reduce_max_3")]; tensor var_2148 = sub(x = x_67_cast_fp16, y = reduce_max_3)[name = string("op_2148")]; tensor var_2154 = exp(x = var_2148)[name = string("op_2154")]; tensor var_2164_axes_0 = const()[name = string("op_2164_axes_0"), val = tensor([-1])]; bool var_2164_keep_dims_0 = const()[name = string("op_2164_keep_dims_0"), val = bool(true)]; tensor var_2164 = reduce_sum(axes = var_2164_axes_0, keep_dims = var_2164_keep_dims_0, x = var_2154)[name = string("op_2164")]; tensor var_2170_cast_fp16 = real_div(x = var_2154, y = var_2164)[name = string("op_2170_cast_fp16")]; bool attn_output_13_transpose_x_0 = const()[name = string("attn_output_13_transpose_x_0"), val = bool(false)]; bool attn_output_13_transpose_y_0 = const()[name = string("attn_output_13_transpose_y_0"), val = bool(false)]; tensor V_expanded_7_cast_fp16 = transpose(perm = V_expanded_7_perm_0, x = reshape_15_cast_fp16)[name = string("transpose_41")]; tensor attn_output_13_cast_fp16 = matmul(transpose_x = attn_output_13_transpose_x_0, transpose_y = attn_output_13_transpose_y_0, x = var_2170_cast_fp16, y = V_expanded_7_cast_fp16)[name = string("attn_output_13_cast_fp16")]; tensor var_2181 = const()[name = string("op_2181"), val = tensor([0, 2, 1, 3])]; tensor var_2188 = const()[name = string("op_2188"), val = tensor([1, 8, -1])]; tensor var_2182_cast_fp16 = transpose(perm = var_2181, x = attn_output_13_cast_fp16)[name = string("transpose_40")]; tensor input_107_cast_fp16 = reshape(shape = var_2188, x = var_2182_cast_fp16)[name = string("input_107_cast_fp16")]; tensor layers_3_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(144978624))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146551552))))[name = string("layers_3_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_31_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_3_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_107_cast_fp16)[name = string("linear_31_cast_fp16")]; int32 var_2197 = const()[name = string("op_2197"), val = int32(-1)]; fp16 const_44_promoted_to_fp16 = const()[name = string("const_44_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2199_cast_fp16 = mul(x = linear_31_cast_fp16, y = const_44_promoted_to_fp16)[name = string("op_2199_cast_fp16")]; bool input_109_interleave_0 = const()[name = string("input_109_interleave_0"), val = bool(false)]; tensor input_109_cast_fp16 = concat(axis = var_2197, interleave = input_109_interleave_0, values = (linear_31_cast_fp16, var_2199_cast_fp16))[name = string("input_109_cast_fp16")]; tensor normed_101_axes_0 = const()[name = string("normed_101_axes_0"), val = tensor([-1])]; fp16 var_2194_to_fp16 = const()[name = string("op_2194_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_101_cast_fp16 = layer_norm(axes = normed_101_axes_0, epsilon = var_2194_to_fp16, x = input_109_cast_fp16)[name = string("normed_101_cast_fp16")]; tensor var_2204_split_sizes_0 = const()[name = string("op_2204_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2204_axis_0 = const()[name = string("op_2204_axis_0"), val = int32(-1)]; tensor var_2204_cast_fp16_0, tensor var_2204_cast_fp16_1 = split(axis = var_2204_axis_0, split_sizes = var_2204_split_sizes_0, x = normed_101_cast_fp16)[name = string("op_2204_cast_fp16")]; tensor layers_3_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_3_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146553152)))]; tensor attn_output_15_cast_fp16 = mul(x = var_2204_cast_fp16_0, y = layers_3_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_15_cast_fp16")]; tensor x_73_cast_fp16 = add(x = x_59_cast_fp16, y = attn_output_15_cast_fp16)[name = string("x_73_cast_fp16")]; int32 var_2213 = const()[name = string("op_2213"), val = int32(-1)]; fp16 const_45_promoted_to_fp16 = const()[name = string("const_45_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2215_cast_fp16 = mul(x = x_73_cast_fp16, y = const_45_promoted_to_fp16)[name = string("op_2215_cast_fp16")]; bool input_111_interleave_0 = const()[name = string("input_111_interleave_0"), val = bool(false)]; tensor input_111_cast_fp16 = concat(axis = var_2213, interleave = input_111_interleave_0, values = (x_73_cast_fp16, var_2215_cast_fp16))[name = string("input_111_cast_fp16")]; tensor normed_105_axes_0 = const()[name = string("normed_105_axes_0"), val = tensor([-1])]; fp16 var_2210_to_fp16 = const()[name = string("op_2210_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_105_cast_fp16 = layer_norm(axes = normed_105_axes_0, epsilon = var_2210_to_fp16, x = input_111_cast_fp16)[name = string("normed_105_cast_fp16")]; tensor var_2220_split_sizes_0 = const()[name = string("op_2220_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2220_axis_0 = const()[name = string("op_2220_axis_0"), val = int32(-1)]; tensor var_2220_cast_fp16_0, tensor var_2220_cast_fp16_1 = split(axis = var_2220_axis_0, split_sizes = var_2220_split_sizes_0, x = normed_105_cast_fp16)[name = string("op_2220_cast_fp16")]; tensor layers_3_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_3_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146556288)))]; tensor var_2222_cast_fp16 = mul(x = var_2220_cast_fp16_0, y = layers_3_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("op_2222_cast_fp16")]; tensor gate_13 = linear(bias = linear_5_bias_0, weight = layers_3_mlp_gate_proj_weight_palettized, x = var_2222_cast_fp16)[name = string("linear_32")]; tensor up_7 = linear(bias = linear_5_bias_0, weight = layers_3_mlp_up_proj_weight_palettized, x = var_2222_cast_fp16)[name = string("linear_33")]; string gate_15_mode_0 = const()[name = string("gate_15_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_15 = gelu(mode = gate_15_mode_0, x = gate_13)[name = string("gate_15")]; tensor input_115 = mul(x = gate_15, y = up_7)[name = string("input_115")]; tensor x_75 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_3_mlp_down_proj_weight_palettized, x = input_115)[name = string("linear_34")]; int32 var_2244 = const()[name = string("op_2244"), val = int32(-1)]; fp16 const_46_promoted = const()[name = string("const_46_promoted"), val = fp16(-0x1p+0)]; tensor var_2246 = mul(x = x_75, y = const_46_promoted)[name = string("op_2246")]; bool input_117_interleave_0 = const()[name = string("input_117_interleave_0"), val = bool(false)]; tensor input_117 = concat(axis = var_2244, interleave = input_117_interleave_0, values = (x_75, var_2246))[name = string("input_117")]; tensor normed_109_axes_0 = const()[name = string("normed_109_axes_0"), val = tensor([-1])]; fp16 var_2241_to_fp16 = const()[name = string("op_2241_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_109_cast_fp16 = layer_norm(axes = normed_109_axes_0, epsilon = var_2241_to_fp16, x = input_117)[name = string("normed_109_cast_fp16")]; tensor var_2251_split_sizes_0 = const()[name = string("op_2251_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2251_axis_0 = const()[name = string("op_2251_axis_0"), val = int32(-1)]; tensor var_2251_0, tensor var_2251_1 = split(axis = var_2251_axis_0, split_sizes = var_2251_split_sizes_0, x = normed_109_cast_fp16)[name = string("op_2251")]; tensor hidden_states_27 = mul(x = var_2251_0, y = layers_3_post_feedforward_layernorm_weight)[name = string("hidden_states_27")]; tensor hidden_states_29_cast_fp16 = add(x = x_73_cast_fp16, y = hidden_states_27)[name = string("hidden_states_29_cast_fp16")]; tensor per_layer_slice_7_begin_0 = const()[name = string("per_layer_slice_7_begin_0"), val = tensor([0, 0, 768])]; tensor per_layer_slice_7_end_0 = const()[name = string("per_layer_slice_7_end_0"), val = tensor([1, 8, 1024])]; tensor per_layer_slice_7_end_mask_0 = const()[name = string("per_layer_slice_7_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_7_cast_fp16 = slice_by_index(begin = per_layer_slice_7_begin_0, end = per_layer_slice_7_end_0, end_mask = per_layer_slice_7_end_mask_0, x = per_layer_combined_out)[name = string("per_layer_slice_7_cast_fp16")]; tensor gated_13 = linear(bias = linear_2_bias_0, weight = layers_3_per_layer_input_gate_weight_palettized, x = hidden_states_29_cast_fp16)[name = string("linear_35")]; string gated_15_mode_0 = const()[name = string("gated_15_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_15 = gelu(mode = gated_15_mode_0, x = gated_13)[name = string("gated_15")]; tensor input_121_cast_fp16 = mul(x = gated_15, y = per_layer_slice_7_cast_fp16)[name = string("input_121_cast_fp16")]; tensor layers_3_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146559424))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146756096))))[name = string("layers_3_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_36_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_3_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_121_cast_fp16)[name = string("linear_36_cast_fp16")]; int32 var_2289 = const()[name = string("op_2289"), val = int32(-1)]; fp16 const_47_promoted_to_fp16 = const()[name = string("const_47_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2291_cast_fp16 = mul(x = linear_36_cast_fp16, y = const_47_promoted_to_fp16)[name = string("op_2291_cast_fp16")]; bool input_123_interleave_0 = const()[name = string("input_123_interleave_0"), val = bool(false)]; tensor input_123_cast_fp16 = concat(axis = var_2289, interleave = input_123_interleave_0, values = (linear_36_cast_fp16, var_2291_cast_fp16))[name = string("input_123_cast_fp16")]; tensor normed_113_axes_0 = const()[name = string("normed_113_axes_0"), val = tensor([-1])]; fp16 var_2286_to_fp16 = const()[name = string("op_2286_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_113_cast_fp16 = layer_norm(axes = normed_113_axes_0, epsilon = var_2286_to_fp16, x = input_123_cast_fp16)[name = string("normed_113_cast_fp16")]; tensor var_2296_split_sizes_0 = const()[name = string("op_2296_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2296_axis_0 = const()[name = string("op_2296_axis_0"), val = int32(-1)]; tensor var_2296_cast_fp16_0, tensor var_2296_cast_fp16_1 = split(axis = var_2296_axis_0, split_sizes = var_2296_split_sizes_0, x = normed_113_cast_fp16)[name = string("op_2296_cast_fp16")]; tensor layers_3_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_3_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146757696)))]; tensor hidden_states_31_cast_fp16 = mul(x = var_2296_cast_fp16_0, y = layers_3_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_31_cast_fp16")]; tensor hidden_states_33_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = hidden_states_31_cast_fp16)[name = string("hidden_states_33_cast_fp16")]; tensor const_48_promoted_to_fp16 = const()[name = string("const_48_promoted_to_fp16"), val = tensor([0x1.26p-2])]; tensor x_79_cast_fp16 = mul(x = hidden_states_33_cast_fp16, y = const_48_promoted_to_fp16)[name = string("x_79_cast_fp16")]; int32 var_2311 = const()[name = string("op_2311"), val = int32(-1)]; fp16 const_49_promoted_to_fp16 = const()[name = string("const_49_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2313_cast_fp16 = mul(x = x_79_cast_fp16, y = const_49_promoted_to_fp16)[name = string("op_2313_cast_fp16")]; bool input_125_interleave_0 = const()[name = string("input_125_interleave_0"), val = bool(false)]; tensor input_125_cast_fp16 = concat(axis = var_2311, interleave = input_125_interleave_0, values = (x_79_cast_fp16, var_2313_cast_fp16))[name = string("input_125_cast_fp16")]; tensor normed_117_axes_0 = const()[name = string("normed_117_axes_0"), val = tensor([-1])]; fp16 var_2308_to_fp16 = const()[name = string("op_2308_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_117_cast_fp16 = layer_norm(axes = normed_117_axes_0, epsilon = var_2308_to_fp16, x = input_125_cast_fp16)[name = string("normed_117_cast_fp16")]; tensor var_2318_split_sizes_0 = const()[name = string("op_2318_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2318_axis_0 = const()[name = string("op_2318_axis_0"), val = int32(-1)]; tensor var_2318_cast_fp16_0, tensor var_2318_cast_fp16_1 = split(axis = var_2318_axis_0, split_sizes = var_2318_split_sizes_0, x = normed_117_cast_fp16)[name = string("op_2318_cast_fp16")]; tensor layers_4_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_4_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146760832)))]; tensor var_2320_cast_fp16 = mul(x = var_2318_cast_fp16_0, y = layers_4_input_layernorm_weight_promoted_to_fp16)[name = string("op_2320_cast_fp16")]; tensor linear_37_bias_0 = const()[name = string("linear_37_bias_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146763968)))]; tensor var_2328 = linear(bias = linear_37_bias_0, weight = layers_4_self_attn_q_proj_weight_palettized, x = var_2320_cast_fp16)[name = string("linear_37")]; tensor var_2333 = const()[name = string("op_2333"), val = tensor([1, 8, 8, 512])]; tensor var_2334 = reshape(shape = var_2333, x = var_2328)[name = string("op_2334")]; tensor var_2339 = const()[name = string("op_2339"), val = tensor([0, 2, 1, 3])]; int32 var_2356 = const()[name = string("op_2356"), val = int32(-1)]; fp16 const_50_promoted = const()[name = string("const_50_promoted"), val = fp16(-0x1p+0)]; tensor var_2340 = transpose(perm = var_2339, x = var_2334)[name = string("transpose_39")]; tensor var_2358 = mul(x = var_2340, y = const_50_promoted)[name = string("op_2358")]; bool input_129_interleave_0 = const()[name = string("input_129_interleave_0"), val = bool(false)]; tensor input_129 = concat(axis = var_2356, interleave = input_129_interleave_0, values = (var_2340, var_2358))[name = string("input_129")]; tensor normed_121_axes_0 = const()[name = string("normed_121_axes_0"), val = tensor([-1])]; fp16 var_2353_to_fp16 = const()[name = string("op_2353_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_121_cast_fp16 = layer_norm(axes = normed_121_axes_0, epsilon = var_2353_to_fp16, x = input_129)[name = string("normed_121_cast_fp16")]; tensor var_2363_split_sizes_0 = const()[name = string("op_2363_split_sizes_0"), val = tensor([512, 512])]; int32 var_2363_axis_0 = const()[name = string("op_2363_axis_0"), val = int32(-1)]; tensor var_2363_0, tensor var_2363_1 = split(axis = var_2363_axis_0, split_sizes = var_2363_split_sizes_0, x = normed_121_cast_fp16)[name = string("op_2363")]; tensor q_35 = mul(x = var_2363_0, y = layers_4_self_attn_q_norm_weight)[name = string("q_35")]; tensor var_2366_cast_fp16 = mul(x = q_35, y = cos_f)[name = string("op_2366_cast_fp16")]; tensor var_2367_split_sizes_0 = const()[name = string("op_2367_split_sizes_0"), val = tensor([256, 256])]; int32 var_2367_axis_0 = const()[name = string("op_2367_axis_0"), val = int32(-1)]; tensor var_2367_0, tensor var_2367_1 = split(axis = var_2367_axis_0, split_sizes = var_2367_split_sizes_0, x = q_35)[name = string("op_2367")]; fp16 const_51_promoted = const()[name = string("const_51_promoted"), val = fp16(-0x1p+0)]; tensor var_2369 = mul(x = var_2367_1, y = const_51_promoted)[name = string("op_2369")]; int32 var_2371 = const()[name = string("op_2371"), val = int32(-1)]; bool var_2372_interleave_0 = const()[name = string("op_2372_interleave_0"), val = bool(false)]; tensor var_2372 = concat(axis = var_2371, interleave = var_2372_interleave_0, values = (var_2369, var_2367_0))[name = string("op_2372")]; tensor var_2373_cast_fp16 = mul(x = var_2372, y = sin_f)[name = string("op_2373_cast_fp16")]; tensor q_39_cast_fp16 = add(x = var_2366_cast_fp16, y = var_2373_cast_fp16)[name = string("q_39_cast_fp16")]; tensor linear_38_bias_0 = const()[name = string("linear_38_bias_0"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146772224)))]; tensor var_2378 = linear(bias = linear_38_bias_0, weight = layers_4_self_attn_k_proj_weight_palettized, x = var_2320_cast_fp16)[name = string("linear_38")]; tensor var_2383 = const()[name = string("op_2383"), val = tensor([1, 8, 1, 512])]; tensor var_2384 = reshape(shape = var_2383, x = var_2378)[name = string("op_2384")]; tensor var_2389 = const()[name = string("op_2389"), val = tensor([0, 2, 1, 3])]; tensor var_2398 = linear(bias = linear_38_bias_0, weight = layers_4_self_attn_v_proj_weight_palettized, x = var_2320_cast_fp16)[name = string("linear_39")]; tensor var_2403 = const()[name = string("op_2403"), val = tensor([1, 8, 1, 512])]; tensor var_2404 = reshape(shape = var_2403, x = var_2398)[name = string("op_2404")]; tensor var_2409 = const()[name = string("op_2409"), val = tensor([0, 2, 1, 3])]; int32 var_2426 = const()[name = string("op_2426"), val = int32(-1)]; fp16 const_52_promoted = const()[name = string("const_52_promoted"), val = fp16(-0x1p+0)]; tensor var_2390 = transpose(perm = var_2389, x = var_2384)[name = string("transpose_38")]; tensor var_2428 = mul(x = var_2390, y = const_52_promoted)[name = string("op_2428")]; bool input_131_interleave_0 = const()[name = string("input_131_interleave_0"), val = bool(false)]; tensor input_131 = concat(axis = var_2426, interleave = input_131_interleave_0, values = (var_2390, var_2428))[name = string("input_131")]; tensor normed_125_axes_0 = const()[name = string("normed_125_axes_0"), val = tensor([-1])]; fp16 var_2423_to_fp16 = const()[name = string("op_2423_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_125_cast_fp16 = layer_norm(axes = normed_125_axes_0, epsilon = var_2423_to_fp16, x = input_131)[name = string("normed_125_cast_fp16")]; tensor var_2433_split_sizes_0 = const()[name = string("op_2433_split_sizes_0"), val = tensor([512, 512])]; int32 var_2433_axis_0 = const()[name = string("op_2433_axis_0"), val = int32(-1)]; tensor var_2433_0, tensor var_2433_1 = split(axis = var_2433_axis_0, split_sizes = var_2433_split_sizes_0, x = normed_125_cast_fp16)[name = string("op_2433")]; tensor q_37 = mul(x = var_2433_0, y = layers_4_self_attn_k_norm_weight)[name = string("q_37")]; fp16 var_2436_promoted = const()[name = string("op_2436_promoted"), val = fp16(0x1p+1)]; tensor var_2410 = transpose(perm = var_2409, x = var_2404)[name = string("transpose_37")]; tensor var_2437 = pow(x = var_2410, y = var_2436_promoted)[name = string("op_2437")]; tensor var_2442_axes_0 = const()[name = string("op_2442_axes_0"), val = tensor([-1])]; bool var_2442_keep_dims_0 = const()[name = string("op_2442_keep_dims_0"), val = bool(true)]; tensor var_2442 = reduce_mean(axes = var_2442_axes_0, keep_dims = var_2442_keep_dims_0, x = var_2437)[name = string("op_2442")]; fp16 var_2444_to_fp16 = const()[name = string("op_2444_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_9_cast_fp16 = add(x = var_2442, y = var_2444_to_fp16)[name = string("mean_sq_9_cast_fp16")]; fp32 var_2446_epsilon_0 = const()[name = string("op_2446_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_2446_cast_fp16 = rsqrt(epsilon = var_2446_epsilon_0, x = mean_sq_9_cast_fp16)[name = string("op_2446_cast_fp16")]; tensor v_cast_fp16 = mul(x = var_2410, y = var_2446_cast_fp16)[name = string("v_cast_fp16")]; tensor var_2448_cast_fp16 = mul(x = q_37, y = cos_f)[name = string("op_2448_cast_fp16")]; tensor var_2449_split_sizes_0 = const()[name = string("op_2449_split_sizes_0"), val = tensor([256, 256])]; int32 var_2449_axis_0 = const()[name = string("op_2449_axis_0"), val = int32(-1)]; tensor var_2449_0, tensor var_2449_1 = split(axis = var_2449_axis_0, split_sizes = var_2449_split_sizes_0, x = q_37)[name = string("op_2449")]; fp16 const_53_promoted = const()[name = string("const_53_promoted"), val = fp16(-0x1p+0)]; tensor var_2451 = mul(x = var_2449_1, y = const_53_promoted)[name = string("op_2451")]; int32 var_2453 = const()[name = string("op_2453"), val = int32(-1)]; bool var_2454_interleave_0 = const()[name = string("op_2454_interleave_0"), val = bool(false)]; tensor var_2454 = concat(axis = var_2453, interleave = var_2454_interleave_0, values = (var_2451, var_2449_0))[name = string("op_2454")]; tensor var_2455_cast_fp16 = mul(x = var_2454, y = sin_f)[name = string("op_2455_cast_fp16")]; tensor k_11_cast_fp16 = add(x = var_2448_cast_fp16, y = var_2455_cast_fp16)[name = string("k_11_cast_fp16")]; int32 var_2459 = const()[name = string("op_2459"), val = int32(8)]; tensor var_2460 = add(x = current_pos, y = var_2459)[name = string("op_2460")]; tensor read_state_1 = read_state(input = kv_cache_full)[name = string("read_state_1")]; tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([0])]; tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([1])]; int32 concat_50_axis_0 = const()[name = string("concat_50_axis_0"), val = int32(0)]; bool concat_50_interleave_0 = const()[name = string("concat_50_interleave_0"), val = bool(false)]; tensor concat_50 = concat(axis = concat_50_axis_0, interleave = concat_50_interleave_0, values = (expand_dims_48, expand_dims_49, current_pos, expand_dims_51))[name = string("concat_50")]; tensor concat_51_values1_0 = const()[name = string("concat_51_values1_0"), val = tensor([0])]; tensor concat_51_values3_0 = const()[name = string("concat_51_values3_0"), val = tensor([0])]; int32 concat_51_axis_0 = const()[name = string("concat_51_axis_0"), val = int32(0)]; bool concat_51_interleave_0 = const()[name = string("concat_51_interleave_0"), val = bool(false)]; tensor concat_51 = concat(axis = concat_51_axis_0, interleave = concat_51_interleave_0, values = (expand_dims_52, concat_51_values1_0, var_2460, concat_51_values3_0))[name = string("concat_51")]; tensor kv_cache_full_internal_tensor_assign_1_stride_0 = const()[name = string("kv_cache_full_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_full_internal_tensor_assign_1_begin_mask_0 = const()[name = string("kv_cache_full_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_full_internal_tensor_assign_1_end_mask_0 = const()[name = string("kv_cache_full_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_full_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("kv_cache_full_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_full_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_50, begin_mask = kv_cache_full_internal_tensor_assign_1_begin_mask_0, end = concat_51, end_mask = kv_cache_full_internal_tensor_assign_1_end_mask_0, squeeze_mask = kv_cache_full_internal_tensor_assign_1_squeeze_mask_0, stride = kv_cache_full_internal_tensor_assign_1_stride_0, update = k_11_cast_fp16, x = read_state_1)[name = string("kv_cache_full_internal_tensor_assign_1_cast_fp16")]; write_state(data = kv_cache_full_internal_tensor_assign_1_cast_fp16, input = kv_cache_full)[name = string("coreml_update_state_24_write_state")]; tensor coreml_update_state_24 = read_state(input = kv_cache_full)[name = string("coreml_update_state_24")]; tensor expand_dims_54 = const()[name = string("expand_dims_54"), val = tensor([1])]; tensor expand_dims_55 = const()[name = string("expand_dims_55"), val = tensor([0])]; tensor expand_dims_57 = const()[name = string("expand_dims_57"), val = tensor([0])]; tensor expand_dims_58 = const()[name = string("expand_dims_58"), val = tensor([2])]; int32 concat_54_axis_0 = const()[name = string("concat_54_axis_0"), val = int32(0)]; bool concat_54_interleave_0 = const()[name = string("concat_54_interleave_0"), val = bool(false)]; tensor concat_54 = concat(axis = concat_54_axis_0, interleave = concat_54_interleave_0, values = (expand_dims_54, expand_dims_55, current_pos, expand_dims_57))[name = string("concat_54")]; tensor concat_55_values1_0 = const()[name = string("concat_55_values1_0"), val = tensor([0])]; tensor concat_55_values3_0 = const()[name = string("concat_55_values3_0"), val = tensor([0])]; int32 concat_55_axis_0 = const()[name = string("concat_55_axis_0"), val = int32(0)]; bool concat_55_interleave_0 = const()[name = string("concat_55_interleave_0"), val = bool(false)]; tensor concat_55 = concat(axis = concat_55_axis_0, interleave = concat_55_interleave_0, values = (expand_dims_58, concat_55_values1_0, var_2460, concat_55_values3_0))[name = string("concat_55")]; tensor kv_cache_full_internal_tensor_assign_2_stride_0 = const()[name = string("kv_cache_full_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_full_internal_tensor_assign_2_begin_mask_0 = const()[name = string("kv_cache_full_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_full_internal_tensor_assign_2_end_mask_0 = const()[name = string("kv_cache_full_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_full_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("kv_cache_full_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_full_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_54, begin_mask = kv_cache_full_internal_tensor_assign_2_begin_mask_0, end = concat_55, end_mask = kv_cache_full_internal_tensor_assign_2_end_mask_0, squeeze_mask = kv_cache_full_internal_tensor_assign_2_squeeze_mask_0, stride = kv_cache_full_internal_tensor_assign_2_stride_0, update = v_cast_fp16, x = coreml_update_state_24)[name = string("kv_cache_full_internal_tensor_assign_2_cast_fp16")]; write_state(data = kv_cache_full_internal_tensor_assign_2_cast_fp16, input = kv_cache_full)[name = string("coreml_update_state_25_write_state")]; tensor coreml_update_state_25 = read_state(input = kv_cache_full)[name = string("coreml_update_state_25")]; tensor var_2510_begin_0 = const()[name = string("op_2510_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_2510_end_0 = const()[name = string("op_2510_end_0"), val = tensor([1, 1, 2048, 512])]; tensor var_2510_end_mask_0 = const()[name = string("op_2510_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2510_cast_fp16 = slice_by_index(begin = var_2510_begin_0, end = var_2510_end_0, end_mask = var_2510_end_mask_0, x = coreml_update_state_25)[name = string("op_2510_cast_fp16")]; tensor var_2530_begin_0 = const()[name = string("op_2530_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_2530_end_0 = const()[name = string("op_2530_end_0"), val = tensor([1, 1, 2048, 512])]; tensor var_2530_end_mask_0 = const()[name = string("op_2530_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_2530_cast_fp16 = slice_by_index(begin = var_2530_begin_0, end = var_2530_end_0, end_mask = var_2530_end_mask_0, x = coreml_update_state_25)[name = string("op_2530_cast_fp16")]; tensor transpose_16_perm_0 = const()[name = string("transpose_16_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_8_reps_0 = const()[name = string("tile_8_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_16_cast_fp16 = transpose(perm = transpose_16_perm_0, x = var_2510_cast_fp16)[name = string("transpose_36")]; tensor tile_8_cast_fp16 = tile(reps = tile_8_reps_0, x = transpose_16_cast_fp16)[name = string("tile_8_cast_fp16")]; tensor concat_56 = const()[name = string("concat_56"), val = tensor([8, 1, 1, 2048, 512])]; tensor reshape_16_cast_fp16 = reshape(shape = concat_56, x = tile_8_cast_fp16)[name = string("reshape_16_cast_fp16")]; tensor transpose_17_perm_0 = const()[name = string("transpose_17_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_57 = const()[name = string("concat_57"), val = tensor([-1, 1, 2048, 512])]; tensor transpose_17_cast_fp16 = transpose(perm = transpose_17_perm_0, x = reshape_16_cast_fp16)[name = string("transpose_35")]; tensor reshape_17_cast_fp16 = reshape(shape = concat_57, x = transpose_17_cast_fp16)[name = string("reshape_17_cast_fp16")]; tensor transpose_36_perm_0 = const()[name = string("transpose_36_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_18_perm_0 = const()[name = string("transpose_18_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_9_reps_0 = const()[name = string("tile_9_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_18_cast_fp16 = transpose(perm = transpose_18_perm_0, x = var_2530_cast_fp16)[name = string("transpose_34")]; tensor tile_9_cast_fp16 = tile(reps = tile_9_reps_0, x = transpose_18_cast_fp16)[name = string("tile_9_cast_fp16")]; tensor concat_58 = const()[name = string("concat_58"), val = tensor([8, 1, 1, 2048, 512])]; tensor reshape_18_cast_fp16 = reshape(shape = concat_58, x = tile_9_cast_fp16)[name = string("reshape_18_cast_fp16")]; tensor transpose_19_perm_0 = const()[name = string("transpose_19_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_59 = const()[name = string("concat_59"), val = tensor([-1, 1, 2048, 512])]; tensor transpose_19_cast_fp16 = transpose(perm = transpose_19_perm_0, x = reshape_18_cast_fp16)[name = string("transpose_33")]; tensor reshape_19_cast_fp16 = reshape(shape = concat_59, x = transpose_19_cast_fp16)[name = string("reshape_19_cast_fp16")]; tensor V_expanded_9_perm_0 = const()[name = string("V_expanded_9_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_17_transpose_x_0 = const()[name = string("attn_weights_17_transpose_x_0"), val = bool(false)]; bool attn_weights_17_transpose_y_0 = const()[name = string("attn_weights_17_transpose_y_0"), val = bool(false)]; tensor transpose_36_cast_fp16 = transpose(perm = transpose_36_perm_0, x = reshape_17_cast_fp16)[name = string("transpose_32")]; tensor attn_weights_17_cast_fp16 = matmul(transpose_x = attn_weights_17_transpose_x_0, transpose_y = attn_weights_17_transpose_y_0, x = q_39_cast_fp16, y = transpose_36_cast_fp16)[name = string("attn_weights_17_cast_fp16")]; tensor x_87_cast_fp16 = add(x = attn_weights_17_cast_fp16, y = causal_mask_full)[name = string("x_87_cast_fp16")]; tensor reduce_max_4_axes_0 = const()[name = string("reduce_max_4_axes_0"), val = tensor([-1])]; bool reduce_max_4_keep_dims_0 = const()[name = string("reduce_max_4_keep_dims_0"), val = bool(true)]; tensor reduce_max_4 = reduce_max(axes = reduce_max_4_axes_0, keep_dims = reduce_max_4_keep_dims_0, x = x_87_cast_fp16)[name = string("reduce_max_4")]; tensor var_2575 = sub(x = x_87_cast_fp16, y = reduce_max_4)[name = string("op_2575")]; tensor var_2581 = exp(x = var_2575)[name = string("op_2581")]; tensor var_2591_axes_0 = const()[name = string("op_2591_axes_0"), val = tensor([-1])]; bool var_2591_keep_dims_0 = const()[name = string("op_2591_keep_dims_0"), val = bool(true)]; tensor var_2591 = reduce_sum(axes = var_2591_axes_0, keep_dims = var_2591_keep_dims_0, x = var_2581)[name = string("op_2591")]; tensor var_2597_cast_fp16 = real_div(x = var_2581, y = var_2591)[name = string("op_2597_cast_fp16")]; bool attn_output_17_transpose_x_0 = const()[name = string("attn_output_17_transpose_x_0"), val = bool(false)]; bool attn_output_17_transpose_y_0 = const()[name = string("attn_output_17_transpose_y_0"), val = bool(false)]; tensor V_expanded_9_cast_fp16 = transpose(perm = V_expanded_9_perm_0, x = reshape_19_cast_fp16)[name = string("transpose_31")]; tensor attn_output_17_cast_fp16 = matmul(transpose_x = attn_output_17_transpose_x_0, transpose_y = attn_output_17_transpose_y_0, x = var_2597_cast_fp16, y = V_expanded_9_cast_fp16)[name = string("attn_output_17_cast_fp16")]; tensor var_2608 = const()[name = string("op_2608"), val = tensor([0, 2, 1, 3])]; tensor var_2615 = const()[name = string("op_2615"), val = tensor([1, 8, -1])]; tensor var_2609_cast_fp16 = transpose(perm = var_2608, x = attn_output_17_cast_fp16)[name = string("transpose_30")]; tensor input_133_cast_fp16 = reshape(shape = var_2615, x = var_2609_cast_fp16)[name = string("input_133_cast_fp16")]; tensor layers_4_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(146773312))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(149919104))))[name = string("layers_4_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_40_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_4_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_133_cast_fp16)[name = string("linear_40_cast_fp16")]; int32 var_2624 = const()[name = string("op_2624"), val = int32(-1)]; fp16 const_54_promoted_to_fp16 = const()[name = string("const_54_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2626_cast_fp16 = mul(x = linear_40_cast_fp16, y = const_54_promoted_to_fp16)[name = string("op_2626_cast_fp16")]; bool input_135_interleave_0 = const()[name = string("input_135_interleave_0"), val = bool(false)]; tensor input_135_cast_fp16 = concat(axis = var_2624, interleave = input_135_interleave_0, values = (linear_40_cast_fp16, var_2626_cast_fp16))[name = string("input_135_cast_fp16")]; tensor normed_129_axes_0 = const()[name = string("normed_129_axes_0"), val = tensor([-1])]; fp16 var_2621_to_fp16 = const()[name = string("op_2621_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_129_cast_fp16 = layer_norm(axes = normed_129_axes_0, epsilon = var_2621_to_fp16, x = input_135_cast_fp16)[name = string("normed_129_cast_fp16")]; tensor var_2631_split_sizes_0 = const()[name = string("op_2631_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2631_axis_0 = const()[name = string("op_2631_axis_0"), val = int32(-1)]; tensor var_2631_cast_fp16_0, tensor var_2631_cast_fp16_1 = split(axis = var_2631_axis_0, split_sizes = var_2631_split_sizes_0, x = normed_129_cast_fp16)[name = string("op_2631_cast_fp16")]; tensor layers_4_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_4_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(149920704)))]; tensor attn_output_19_cast_fp16 = mul(x = var_2631_cast_fp16_0, y = layers_4_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_19_cast_fp16")]; tensor x_93_cast_fp16 = add(x = x_79_cast_fp16, y = attn_output_19_cast_fp16)[name = string("x_93_cast_fp16")]; int32 var_2640 = const()[name = string("op_2640"), val = int32(-1)]; fp16 const_55_promoted_to_fp16 = const()[name = string("const_55_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2642_cast_fp16 = mul(x = x_93_cast_fp16, y = const_55_promoted_to_fp16)[name = string("op_2642_cast_fp16")]; bool input_137_interleave_0 = const()[name = string("input_137_interleave_0"), val = bool(false)]; tensor input_137_cast_fp16 = concat(axis = var_2640, interleave = input_137_interleave_0, values = (x_93_cast_fp16, var_2642_cast_fp16))[name = string("input_137_cast_fp16")]; tensor normed_133_axes_0 = const()[name = string("normed_133_axes_0"), val = tensor([-1])]; fp16 var_2637_to_fp16 = const()[name = string("op_2637_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_133_cast_fp16 = layer_norm(axes = normed_133_axes_0, epsilon = var_2637_to_fp16, x = input_137_cast_fp16)[name = string("normed_133_cast_fp16")]; tensor var_2647_split_sizes_0 = const()[name = string("op_2647_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2647_axis_0 = const()[name = string("op_2647_axis_0"), val = int32(-1)]; tensor var_2647_cast_fp16_0, tensor var_2647_cast_fp16_1 = split(axis = var_2647_axis_0, split_sizes = var_2647_split_sizes_0, x = normed_133_cast_fp16)[name = string("op_2647_cast_fp16")]; tensor layers_4_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_4_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(149923840)))]; tensor var_2649_cast_fp16 = mul(x = var_2647_cast_fp16_0, y = layers_4_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("op_2649_cast_fp16")]; tensor gate_17 = linear(bias = linear_5_bias_0, weight = layers_4_mlp_gate_proj_weight_palettized, x = var_2649_cast_fp16)[name = string("linear_41")]; tensor up_9 = linear(bias = linear_5_bias_0, weight = layers_4_mlp_up_proj_weight_palettized, x = var_2649_cast_fp16)[name = string("linear_42")]; string gate_19_mode_0 = const()[name = string("gate_19_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_19 = gelu(mode = gate_19_mode_0, x = gate_17)[name = string("gate_19")]; tensor input_141 = mul(x = gate_19, y = up_9)[name = string("input_141")]; tensor x_95 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_4_mlp_down_proj_weight_palettized, x = input_141)[name = string("linear_43")]; int32 var_2671 = const()[name = string("op_2671"), val = int32(-1)]; fp16 const_56_promoted = const()[name = string("const_56_promoted"), val = fp16(-0x1p+0)]; tensor var_2673 = mul(x = x_95, y = const_56_promoted)[name = string("op_2673")]; bool input_143_interleave_0 = const()[name = string("input_143_interleave_0"), val = bool(false)]; tensor input_143 = concat(axis = var_2671, interleave = input_143_interleave_0, values = (x_95, var_2673))[name = string("input_143")]; tensor normed_137_axes_0 = const()[name = string("normed_137_axes_0"), val = tensor([-1])]; fp16 var_2668_to_fp16 = const()[name = string("op_2668_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_137_cast_fp16 = layer_norm(axes = normed_137_axes_0, epsilon = var_2668_to_fp16, x = input_143)[name = string("normed_137_cast_fp16")]; tensor var_2678_split_sizes_0 = const()[name = string("op_2678_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2678_axis_0 = const()[name = string("op_2678_axis_0"), val = int32(-1)]; tensor var_2678_0, tensor var_2678_1 = split(axis = var_2678_axis_0, split_sizes = var_2678_split_sizes_0, x = normed_137_cast_fp16)[name = string("op_2678")]; tensor hidden_states_35 = mul(x = var_2678_0, y = layers_4_post_feedforward_layernorm_weight)[name = string("hidden_states_35")]; tensor hidden_states_37_cast_fp16 = add(x = x_93_cast_fp16, y = hidden_states_35)[name = string("hidden_states_37_cast_fp16")]; tensor per_layer_slice_9_begin_0 = const()[name = string("per_layer_slice_9_begin_0"), val = tensor([0, 0, 1024])]; tensor per_layer_slice_9_end_0 = const()[name = string("per_layer_slice_9_end_0"), val = tensor([1, 8, 1280])]; tensor per_layer_slice_9_end_mask_0 = const()[name = string("per_layer_slice_9_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_9_cast_fp16 = slice_by_index(begin = per_layer_slice_9_begin_0, end = per_layer_slice_9_end_0, end_mask = per_layer_slice_9_end_mask_0, x = per_layer_combined_out)[name = string("per_layer_slice_9_cast_fp16")]; tensor gated_17 = linear(bias = linear_2_bias_0, weight = layers_4_per_layer_input_gate_weight_palettized, x = hidden_states_37_cast_fp16)[name = string("linear_44")]; string gated_19_mode_0 = const()[name = string("gated_19_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_19 = gelu(mode = gated_19_mode_0, x = gated_17)[name = string("gated_19")]; tensor input_147_cast_fp16 = mul(x = gated_19, y = per_layer_slice_9_cast_fp16)[name = string("input_147_cast_fp16")]; tensor layers_4_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(149926976))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(150123648))))[name = string("layers_4_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_45_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_4_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_147_cast_fp16)[name = string("linear_45_cast_fp16")]; int32 var_2716 = const()[name = string("op_2716"), val = int32(-1)]; fp16 const_57_promoted_to_fp16 = const()[name = string("const_57_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2718_cast_fp16 = mul(x = linear_45_cast_fp16, y = const_57_promoted_to_fp16)[name = string("op_2718_cast_fp16")]; bool input_149_interleave_0 = const()[name = string("input_149_interleave_0"), val = bool(false)]; tensor input_149_cast_fp16 = concat(axis = var_2716, interleave = input_149_interleave_0, values = (linear_45_cast_fp16, var_2718_cast_fp16))[name = string("input_149_cast_fp16")]; tensor normed_141_axes_0 = const()[name = string("normed_141_axes_0"), val = tensor([-1])]; fp16 var_2713_to_fp16 = const()[name = string("op_2713_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_141_cast_fp16 = layer_norm(axes = normed_141_axes_0, epsilon = var_2713_to_fp16, x = input_149_cast_fp16)[name = string("normed_141_cast_fp16")]; tensor var_2723_split_sizes_0 = const()[name = string("op_2723_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2723_axis_0 = const()[name = string("op_2723_axis_0"), val = int32(-1)]; tensor var_2723_cast_fp16_0, tensor var_2723_cast_fp16_1 = split(axis = var_2723_axis_0, split_sizes = var_2723_split_sizes_0, x = normed_141_cast_fp16)[name = string("op_2723_cast_fp16")]; tensor layers_4_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_4_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(150125248)))]; tensor hidden_states_39_cast_fp16 = mul(x = var_2723_cast_fp16_0, y = layers_4_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_39_cast_fp16")]; tensor hidden_states_41_cast_fp16 = add(x = hidden_states_37_cast_fp16, y = hidden_states_39_cast_fp16)[name = string("hidden_states_41_cast_fp16")]; tensor const_58_promoted_to_fp16 = const()[name = string("const_58_promoted_to_fp16"), val = tensor([0x1.fep-2])]; tensor x_99_cast_fp16 = mul(x = hidden_states_41_cast_fp16, y = const_58_promoted_to_fp16)[name = string("x_99_cast_fp16")]; int32 var_2738 = const()[name = string("op_2738"), val = int32(-1)]; fp16 const_59_promoted_to_fp16 = const()[name = string("const_59_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2740_cast_fp16 = mul(x = x_99_cast_fp16, y = const_59_promoted_to_fp16)[name = string("op_2740_cast_fp16")]; bool input_151_interleave_0 = const()[name = string("input_151_interleave_0"), val = bool(false)]; tensor input_151_cast_fp16 = concat(axis = var_2738, interleave = input_151_interleave_0, values = (x_99_cast_fp16, var_2740_cast_fp16))[name = string("input_151_cast_fp16")]; tensor normed_145_axes_0 = const()[name = string("normed_145_axes_0"), val = tensor([-1])]; fp16 var_2735_to_fp16 = const()[name = string("op_2735_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_145_cast_fp16 = layer_norm(axes = normed_145_axes_0, epsilon = var_2735_to_fp16, x = input_151_cast_fp16)[name = string("normed_145_cast_fp16")]; tensor var_2745_split_sizes_0 = const()[name = string("op_2745_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_2745_axis_0 = const()[name = string("op_2745_axis_0"), val = int32(-1)]; tensor var_2745_cast_fp16_0, tensor var_2745_cast_fp16_1 = split(axis = var_2745_axis_0, split_sizes = var_2745_split_sizes_0, x = normed_145_cast_fp16)[name = string("op_2745_cast_fp16")]; tensor layers_5_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_5_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(150128384)))]; tensor var_2747_cast_fp16 = mul(x = var_2745_cast_fp16_0, y = layers_5_input_layernorm_weight_promoted_to_fp16)[name = string("op_2747_cast_fp16")]; tensor var_2755 = linear(bias = linear_1_bias_0, weight = layers_5_self_attn_q_proj_weight_palettized, x = var_2747_cast_fp16)[name = string("linear_46")]; tensor var_2760 = const()[name = string("op_2760"), val = tensor([1, 8, 8, 256])]; tensor var_2761 = reshape(shape = var_2760, x = var_2755)[name = string("op_2761")]; tensor var_2766 = const()[name = string("op_2766"), val = tensor([0, 2, 1, 3])]; int32 var_2783 = const()[name = string("op_2783"), val = int32(-1)]; fp16 const_60_promoted = const()[name = string("const_60_promoted"), val = fp16(-0x1p+0)]; tensor var_2767 = transpose(perm = var_2766, x = var_2761)[name = string("transpose_29")]; tensor var_2785 = mul(x = var_2767, y = const_60_promoted)[name = string("op_2785")]; bool input_155_interleave_0 = const()[name = string("input_155_interleave_0"), val = bool(false)]; tensor input_155 = concat(axis = var_2783, interleave = input_155_interleave_0, values = (var_2767, var_2785))[name = string("input_155")]; tensor normed_149_axes_0 = const()[name = string("normed_149_axes_0"), val = tensor([-1])]; fp16 var_2780_to_fp16 = const()[name = string("op_2780_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_149_cast_fp16 = layer_norm(axes = normed_149_axes_0, epsilon = var_2780_to_fp16, x = input_155)[name = string("normed_149_cast_fp16")]; tensor var_2790_split_sizes_0 = const()[name = string("op_2790_split_sizes_0"), val = tensor([256, 256])]; int32 var_2790_axis_0 = const()[name = string("op_2790_axis_0"), val = int32(-1)]; tensor var_2790_0, tensor var_2790_1 = split(axis = var_2790_axis_0, split_sizes = var_2790_split_sizes_0, x = normed_149_cast_fp16)[name = string("op_2790")]; tensor q_43 = mul(x = var_2790_0, y = layers_5_self_attn_q_norm_weight)[name = string("q_43")]; tensor var_2793_cast_fp16 = mul(x = q_43, y = cos_s)[name = string("op_2793_cast_fp16")]; tensor var_2794_split_sizes_0 = const()[name = string("op_2794_split_sizes_0"), val = tensor([128, 128])]; int32 var_2794_axis_0 = const()[name = string("op_2794_axis_0"), val = int32(-1)]; tensor var_2794_0, tensor var_2794_1 = split(axis = var_2794_axis_0, split_sizes = var_2794_split_sizes_0, x = q_43)[name = string("op_2794")]; fp16 const_61_promoted = const()[name = string("const_61_promoted"), val = fp16(-0x1p+0)]; tensor var_2796 = mul(x = var_2794_1, y = const_61_promoted)[name = string("op_2796")]; int32 var_2798 = const()[name = string("op_2798"), val = int32(-1)]; bool var_2799_interleave_0 = const()[name = string("op_2799_interleave_0"), val = bool(false)]; tensor var_2799 = concat(axis = var_2798, interleave = var_2799_interleave_0, values = (var_2796, var_2794_0))[name = string("op_2799")]; tensor var_2800_cast_fp16 = mul(x = var_2799, y = sin_s)[name = string("op_2800_cast_fp16")]; tensor q_47_cast_fp16 = add(x = var_2793_cast_fp16, y = var_2800_cast_fp16)[name = string("q_47_cast_fp16")]; tensor var_2805 = linear(bias = linear_2_bias_0, weight = layers_5_self_attn_k_proj_weight_palettized, x = var_2747_cast_fp16)[name = string("linear_47")]; tensor var_2810 = const()[name = string("op_2810"), val = tensor([1, 8, 1, 256])]; tensor var_2811 = reshape(shape = var_2810, x = var_2805)[name = string("op_2811")]; tensor var_2816 = const()[name = string("op_2816"), val = tensor([0, 2, 1, 3])]; tensor var_2825 = linear(bias = linear_2_bias_0, weight = layers_5_self_attn_v_proj_weight_palettized, x = var_2747_cast_fp16)[name = string("linear_48")]; tensor var_2830 = const()[name = string("op_2830"), val = tensor([1, 8, 1, 256])]; tensor var_2831 = reshape(shape = var_2830, x = var_2825)[name = string("op_2831")]; tensor var_2836 = const()[name = string("op_2836"), val = tensor([0, 2, 1, 3])]; int32 var_2853 = const()[name = string("op_2853"), val = int32(-1)]; fp16 const_62_promoted = const()[name = string("const_62_promoted"), val = fp16(-0x1p+0)]; tensor var_2817 = transpose(perm = var_2816, x = var_2811)[name = string("transpose_28")]; tensor var_2855 = mul(x = var_2817, y = const_62_promoted)[name = string("op_2855")]; bool input_157_interleave_0 = const()[name = string("input_157_interleave_0"), val = bool(false)]; tensor input_157 = concat(axis = var_2853, interleave = input_157_interleave_0, values = (var_2817, var_2855))[name = string("input_157")]; tensor normed_153_axes_0 = const()[name = string("normed_153_axes_0"), val = tensor([-1])]; fp16 var_2850_to_fp16 = const()[name = string("op_2850_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_153_cast_fp16 = layer_norm(axes = normed_153_axes_0, epsilon = var_2850_to_fp16, x = input_157)[name = string("normed_153_cast_fp16")]; tensor var_2860_split_sizes_0 = const()[name = string("op_2860_split_sizes_0"), val = tensor([256, 256])]; int32 var_2860_axis_0 = const()[name = string("op_2860_axis_0"), val = int32(-1)]; tensor var_2860_0, tensor var_2860_1 = split(axis = var_2860_axis_0, split_sizes = var_2860_split_sizes_0, x = normed_153_cast_fp16)[name = string("op_2860")]; tensor q_45 = mul(x = var_2860_0, y = layers_0_self_attn_k_norm_weight)[name = string("q_45")]; fp16 var_2863_promoted = const()[name = string("op_2863_promoted"), val = fp16(0x1p+1)]; tensor var_2837 = transpose(perm = var_2836, x = var_2831)[name = string("transpose_27")]; tensor var_2864 = pow(x = var_2837, y = var_2863_promoted)[name = string("op_2864")]; tensor var_2869_axes_0 = const()[name = string("op_2869_axes_0"), val = tensor([-1])]; bool var_2869_keep_dims_0 = const()[name = string("op_2869_keep_dims_0"), val = bool(true)]; tensor var_2869 = reduce_mean(axes = var_2869_axes_0, keep_dims = var_2869_keep_dims_0, x = var_2864)[name = string("op_2869")]; fp16 var_2871_to_fp16 = const()[name = string("op_2871_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_11_cast_fp16 = add(x = var_2869, y = var_2871_to_fp16)[name = string("mean_sq_11_cast_fp16")]; fp32 var_2873_epsilon_0 = const()[name = string("op_2873_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_2873_cast_fp16 = rsqrt(epsilon = var_2873_epsilon_0, x = mean_sq_11_cast_fp16)[name = string("op_2873_cast_fp16")]; tensor input_161_cast_fp16 = mul(x = var_2837, y = var_2873_cast_fp16)[name = string("input_161_cast_fp16")]; tensor var_2875_cast_fp16 = mul(x = q_45, y = cos_s)[name = string("op_2875_cast_fp16")]; tensor var_2876_split_sizes_0 = const()[name = string("op_2876_split_sizes_0"), val = tensor([128, 128])]; int32 var_2876_axis_0 = const()[name = string("op_2876_axis_0"), val = int32(-1)]; tensor var_2876_0, tensor var_2876_1 = split(axis = var_2876_axis_0, split_sizes = var_2876_split_sizes_0, x = q_45)[name = string("op_2876")]; fp16 const_63_promoted = const()[name = string("const_63_promoted"), val = fp16(-0x1p+0)]; tensor var_2878 = mul(x = var_2876_1, y = const_63_promoted)[name = string("op_2878")]; int32 var_2880 = const()[name = string("op_2880"), val = int32(-1)]; bool var_2881_interleave_0 = const()[name = string("op_2881_interleave_0"), val = bool(false)]; tensor var_2881 = concat(axis = var_2880, interleave = var_2881_interleave_0, values = (var_2878, var_2876_0))[name = string("op_2881")]; tensor var_2882_cast_fp16 = mul(x = var_2881, y = sin_s)[name = string("op_2882_cast_fp16")]; tensor input_159_cast_fp16 = add(x = var_2875_cast_fp16, y = var_2882_cast_fp16)[name = string("input_159_cast_fp16")]; tensor k_padded_9_pad_0 = const()[name = string("k_padded_9_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_9_mode_0 = const()[name = string("k_padded_9_mode_0"), val = string("constant")]; fp16 const_64_to_fp16 = const()[name = string("const_64_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_9_cast_fp16 = pad(constant_val = const_64_to_fp16, mode = k_padded_9_mode_0, pad = k_padded_9_pad_0, x = input_159_cast_fp16)[name = string("k_padded_9_cast_fp16")]; tensor v_padded_9_pad_0 = const()[name = string("v_padded_9_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_9_mode_0 = const()[name = string("v_padded_9_mode_0"), val = string("constant")]; fp16 const_65_to_fp16 = const()[name = string("const_65_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_9_cast_fp16 = pad(constant_val = const_65_to_fp16, mode = v_padded_9_mode_0, pad = v_padded_9_pad_0, x = input_161_cast_fp16)[name = string("v_padded_9_cast_fp16")]; tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([8])]; tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([9])]; int32 concat_62_axis_0 = const()[name = string("concat_62_axis_0"), val = int32(0)]; bool concat_62_interleave_0 = const()[name = string("concat_62_interleave_0"), val = bool(false)]; tensor concat_62 = concat(axis = concat_62_axis_0, interleave = concat_62_interleave_0, values = (expand_dims_60, expand_dims_61, ring_pos, expand_dims_63))[name = string("concat_62")]; tensor concat_63_values1_0 = const()[name = string("concat_63_values1_0"), val = tensor([0])]; tensor concat_63_values3_0 = const()[name = string("concat_63_values3_0"), val = tensor([0])]; int32 concat_63_axis_0 = const()[name = string("concat_63_axis_0"), val = int32(0)]; bool concat_63_interleave_0 = const()[name = string("concat_63_interleave_0"), val = bool(false)]; tensor concat_63 = concat(axis = concat_63_axis_0, interleave = concat_63_interleave_0, values = (expand_dims_64, concat_63_values1_0, var_716, concat_63_values3_0))[name = string("concat_63")]; tensor kv_cache_sliding_internal_tensor_assign_9_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_9_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_9_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_62, begin_mask = kv_cache_sliding_internal_tensor_assign_9_begin_mask_0, end = concat_63, end_mask = kv_cache_sliding_internal_tensor_assign_9_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_9_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_9_stride_0, update = k_padded_9_cast_fp16, x = coreml_update_state_23)[name = string("kv_cache_sliding_internal_tensor_assign_9_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_9_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_26_write_state")]; tensor coreml_update_state_26 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_26")]; tensor expand_dims_66 = const()[name = string("expand_dims_66"), val = tensor([9])]; tensor expand_dims_67 = const()[name = string("expand_dims_67"), val = tensor([0])]; tensor expand_dims_69 = const()[name = string("expand_dims_69"), val = tensor([0])]; tensor expand_dims_70 = const()[name = string("expand_dims_70"), val = tensor([10])]; int32 concat_66_axis_0 = const()[name = string("concat_66_axis_0"), val = int32(0)]; bool concat_66_interleave_0 = const()[name = string("concat_66_interleave_0"), val = bool(false)]; tensor concat_66 = concat(axis = concat_66_axis_0, interleave = concat_66_interleave_0, values = (expand_dims_66, expand_dims_67, ring_pos, expand_dims_69))[name = string("concat_66")]; tensor concat_67_values1_0 = const()[name = string("concat_67_values1_0"), val = tensor([0])]; tensor concat_67_values3_0 = const()[name = string("concat_67_values3_0"), val = tensor([0])]; int32 concat_67_axis_0 = const()[name = string("concat_67_axis_0"), val = int32(0)]; bool concat_67_interleave_0 = const()[name = string("concat_67_interleave_0"), val = bool(false)]; tensor concat_67 = concat(axis = concat_67_axis_0, interleave = concat_67_interleave_0, values = (expand_dims_70, concat_67_values1_0, var_716, concat_67_values3_0))[name = string("concat_67")]; tensor kv_cache_sliding_internal_tensor_assign_10_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_10_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_10_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_66, begin_mask = kv_cache_sliding_internal_tensor_assign_10_begin_mask_0, end = concat_67, end_mask = kv_cache_sliding_internal_tensor_assign_10_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_10_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_10_stride_0, update = v_padded_9_cast_fp16, x = coreml_update_state_26)[name = string("kv_cache_sliding_internal_tensor_assign_10_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_10_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_27_write_state")]; tensor coreml_update_state_27 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_27")]; tensor var_2949_begin_0 = const()[name = string("op_2949_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_2949_end_0 = const()[name = string("op_2949_end_0"), val = tensor([9, 1, 512, 512])]; tensor var_2949_end_mask_0 = const()[name = string("op_2949_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2949_cast_fp16 = slice_by_index(begin = var_2949_begin_0, end = var_2949_end_0, end_mask = var_2949_end_mask_0, x = coreml_update_state_27)[name = string("op_2949_cast_fp16")]; tensor K_sliding_slice_9_begin_0 = const()[name = string("K_sliding_slice_9_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_sliding_slice_9_end_0 = const()[name = string("K_sliding_slice_9_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_sliding_slice_9_end_mask_0 = const()[name = string("K_sliding_slice_9_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_sliding_slice_9_cast_fp16 = slice_by_index(begin = K_sliding_slice_9_begin_0, end = K_sliding_slice_9_end_0, end_mask = K_sliding_slice_9_end_mask_0, x = var_2949_cast_fp16)[name = string("K_sliding_slice_9_cast_fp16")]; tensor var_2969_begin_0 = const()[name = string("op_2969_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_2969_end_0 = const()[name = string("op_2969_end_0"), val = tensor([10, 1, 512, 512])]; tensor var_2969_end_mask_0 = const()[name = string("op_2969_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2969_cast_fp16 = slice_by_index(begin = var_2969_begin_0, end = var_2969_end_0, end_mask = var_2969_end_mask_0, x = coreml_update_state_27)[name = string("op_2969_cast_fp16")]; tensor V_for_attn_9_begin_0 = const()[name = string("V_for_attn_9_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_9_end_0 = const()[name = string("V_for_attn_9_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_9_end_mask_0 = const()[name = string("V_for_attn_9_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_9_cast_fp16 = slice_by_index(begin = V_for_attn_9_begin_0, end = V_for_attn_9_end_0, end_mask = V_for_attn_9_end_mask_0, x = var_2969_cast_fp16)[name = string("V_for_attn_9_cast_fp16")]; tensor transpose_20_perm_0 = const()[name = string("transpose_20_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_10_reps_0 = const()[name = string("tile_10_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_20_cast_fp16 = transpose(perm = transpose_20_perm_0, x = K_sliding_slice_9_cast_fp16)[name = string("transpose_26")]; tensor tile_10_cast_fp16 = tile(reps = tile_10_reps_0, x = transpose_20_cast_fp16)[name = string("tile_10_cast_fp16")]; tensor concat_68 = const()[name = string("concat_68"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_20_cast_fp16 = reshape(shape = concat_68, x = tile_10_cast_fp16)[name = string("reshape_20_cast_fp16")]; tensor transpose_21_perm_0 = const()[name = string("transpose_21_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_69 = const()[name = string("concat_69"), val = tensor([-1, 1, 512, 256])]; tensor transpose_21_cast_fp16 = transpose(perm = transpose_21_perm_0, x = reshape_20_cast_fp16)[name = string("transpose_25")]; tensor reshape_21_cast_fp16 = reshape(shape = concat_69, x = transpose_21_cast_fp16)[name = string("reshape_21_cast_fp16")]; tensor transpose_37_perm_0 = const()[name = string("transpose_37_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_22_perm_0 = const()[name = string("transpose_22_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_11_reps_0 = const()[name = string("tile_11_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_22_cast_fp16 = transpose(perm = transpose_22_perm_0, x = V_for_attn_9_cast_fp16)[name = string("transpose_24")]; tensor tile_11_cast_fp16 = tile(reps = tile_11_reps_0, x = transpose_22_cast_fp16)[name = string("tile_11_cast_fp16")]; tensor concat_70 = const()[name = string("concat_70"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_22_cast_fp16 = reshape(shape = concat_70, x = tile_11_cast_fp16)[name = string("reshape_22_cast_fp16")]; tensor transpose_23_perm_0 = const()[name = string("transpose_23_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_71 = const()[name = string("concat_71"), val = tensor([-1, 1, 512, 256])]; tensor transpose_23_cast_fp16 = transpose(perm = transpose_23_perm_0, x = reshape_22_cast_fp16)[name = string("transpose_23")]; tensor reshape_23_cast_fp16 = reshape(shape = concat_71, x = transpose_23_cast_fp16)[name = string("reshape_23_cast_fp16")]; tensor V_expanded_11_perm_0 = const()[name = string("V_expanded_11_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_21_transpose_x_0 = const()[name = string("attn_weights_21_transpose_x_0"), val = bool(false)]; bool attn_weights_21_transpose_y_0 = const()[name = string("attn_weights_21_transpose_y_0"), val = bool(false)]; tensor transpose_37_cast_fp16 = transpose(perm = transpose_37_perm_0, x = reshape_21_cast_fp16)[name = string("transpose_22")]; tensor attn_weights_21_cast_fp16 = matmul(transpose_x = attn_weights_21_transpose_x_0, transpose_y = attn_weights_21_transpose_y_0, x = q_47_cast_fp16, y = transpose_37_cast_fp16)[name = string("attn_weights_21_cast_fp16")]; tensor x_107_cast_fp16 = add(x = attn_weights_21_cast_fp16, y = causal_mask_sliding)[name = string("x_107_cast_fp16")]; tensor reduce_max_5_axes_0 = const()[name = string("reduce_max_5_axes_0"), val = tensor([-1])]; bool reduce_max_5_keep_dims_0 = const()[name = string("reduce_max_5_keep_dims_0"), val = bool(true)]; tensor reduce_max_5 = reduce_max(axes = reduce_max_5_axes_0, keep_dims = reduce_max_5_keep_dims_0, x = x_107_cast_fp16)[name = string("reduce_max_5")]; tensor var_3014 = sub(x = x_107_cast_fp16, y = reduce_max_5)[name = string("op_3014")]; tensor var_3020 = exp(x = var_3014)[name = string("op_3020")]; tensor var_3030_axes_0 = const()[name = string("op_3030_axes_0"), val = tensor([-1])]; bool var_3030_keep_dims_0 = const()[name = string("op_3030_keep_dims_0"), val = bool(true)]; tensor var_3030 = reduce_sum(axes = var_3030_axes_0, keep_dims = var_3030_keep_dims_0, x = var_3020)[name = string("op_3030")]; tensor var_3036_cast_fp16 = real_div(x = var_3020, y = var_3030)[name = string("op_3036_cast_fp16")]; bool attn_output_21_transpose_x_0 = const()[name = string("attn_output_21_transpose_x_0"), val = bool(false)]; bool attn_output_21_transpose_y_0 = const()[name = string("attn_output_21_transpose_y_0"), val = bool(false)]; tensor V_expanded_11_cast_fp16 = transpose(perm = V_expanded_11_perm_0, x = reshape_23_cast_fp16)[name = string("transpose_21")]; tensor attn_output_21_cast_fp16 = matmul(transpose_x = attn_output_21_transpose_x_0, transpose_y = attn_output_21_transpose_y_0, x = var_3036_cast_fp16, y = V_expanded_11_cast_fp16)[name = string("attn_output_21_cast_fp16")]; tensor var_3047 = const()[name = string("op_3047"), val = tensor([0, 2, 1, 3])]; tensor var_3054 = const()[name = string("op_3054"), val = tensor([1, 8, -1])]; tensor var_3048_cast_fp16 = transpose(perm = var_3047, x = attn_output_21_cast_fp16)[name = string("transpose_20")]; tensor input_163_cast_fp16 = reshape(shape = var_3054, x = var_3048_cast_fp16)[name = string("input_163_cast_fp16")]; tensor layers_5_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(150131520))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151704448))))[name = string("layers_5_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_49_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_5_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_163_cast_fp16)[name = string("linear_49_cast_fp16")]; int32 var_3063 = const()[name = string("op_3063"), val = int32(-1)]; fp16 const_66_promoted_to_fp16 = const()[name = string("const_66_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3065_cast_fp16 = mul(x = linear_49_cast_fp16, y = const_66_promoted_to_fp16)[name = string("op_3065_cast_fp16")]; bool input_165_interleave_0 = const()[name = string("input_165_interleave_0"), val = bool(false)]; tensor input_165_cast_fp16 = concat(axis = var_3063, interleave = input_165_interleave_0, values = (linear_49_cast_fp16, var_3065_cast_fp16))[name = string("input_165_cast_fp16")]; tensor normed_157_axes_0 = const()[name = string("normed_157_axes_0"), val = tensor([-1])]; fp16 var_3060_to_fp16 = const()[name = string("op_3060_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_157_cast_fp16 = layer_norm(axes = normed_157_axes_0, epsilon = var_3060_to_fp16, x = input_165_cast_fp16)[name = string("normed_157_cast_fp16")]; tensor var_3070_split_sizes_0 = const()[name = string("op_3070_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3070_axis_0 = const()[name = string("op_3070_axis_0"), val = int32(-1)]; tensor var_3070_cast_fp16_0, tensor var_3070_cast_fp16_1 = split(axis = var_3070_axis_0, split_sizes = var_3070_split_sizes_0, x = normed_157_cast_fp16)[name = string("op_3070_cast_fp16")]; tensor layers_5_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_5_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151706048)))]; tensor attn_output_23_cast_fp16 = mul(x = var_3070_cast_fp16_0, y = layers_5_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_23_cast_fp16")]; tensor x_113_cast_fp16 = add(x = x_99_cast_fp16, y = attn_output_23_cast_fp16)[name = string("x_113_cast_fp16")]; int32 var_3079 = const()[name = string("op_3079"), val = int32(-1)]; fp16 const_67_promoted_to_fp16 = const()[name = string("const_67_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3081_cast_fp16 = mul(x = x_113_cast_fp16, y = const_67_promoted_to_fp16)[name = string("op_3081_cast_fp16")]; bool input_167_interleave_0 = const()[name = string("input_167_interleave_0"), val = bool(false)]; tensor input_167_cast_fp16 = concat(axis = var_3079, interleave = input_167_interleave_0, values = (x_113_cast_fp16, var_3081_cast_fp16))[name = string("input_167_cast_fp16")]; tensor normed_161_axes_0 = const()[name = string("normed_161_axes_0"), val = tensor([-1])]; fp16 var_3076_to_fp16 = const()[name = string("op_3076_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_161_cast_fp16 = layer_norm(axes = normed_161_axes_0, epsilon = var_3076_to_fp16, x = input_167_cast_fp16)[name = string("normed_161_cast_fp16")]; tensor var_3086_split_sizes_0 = const()[name = string("op_3086_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3086_axis_0 = const()[name = string("op_3086_axis_0"), val = int32(-1)]; tensor var_3086_cast_fp16_0, tensor var_3086_cast_fp16_1 = split(axis = var_3086_axis_0, split_sizes = var_3086_split_sizes_0, x = normed_161_cast_fp16)[name = string("op_3086_cast_fp16")]; tensor layers_5_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_5_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151709184)))]; tensor var_3088_cast_fp16 = mul(x = var_3086_cast_fp16_0, y = layers_5_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("op_3088_cast_fp16")]; tensor gate_21 = linear(bias = linear_5_bias_0, weight = layers_5_mlp_gate_proj_weight_palettized, x = var_3088_cast_fp16)[name = string("linear_50")]; tensor up_11 = linear(bias = linear_5_bias_0, weight = layers_5_mlp_up_proj_weight_palettized, x = var_3088_cast_fp16)[name = string("linear_51")]; string gate_23_mode_0 = const()[name = string("gate_23_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_23 = gelu(mode = gate_23_mode_0, x = gate_21)[name = string("gate_23")]; tensor input_171 = mul(x = gate_23, y = up_11)[name = string("input_171")]; tensor x_115 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_5_mlp_down_proj_weight_palettized, x = input_171)[name = string("linear_52")]; int32 var_3110 = const()[name = string("op_3110"), val = int32(-1)]; fp16 const_68_promoted = const()[name = string("const_68_promoted"), val = fp16(-0x1p+0)]; tensor var_3112 = mul(x = x_115, y = const_68_promoted)[name = string("op_3112")]; bool input_173_interleave_0 = const()[name = string("input_173_interleave_0"), val = bool(false)]; tensor input_173 = concat(axis = var_3110, interleave = input_173_interleave_0, values = (x_115, var_3112))[name = string("input_173")]; tensor normed_165_axes_0 = const()[name = string("normed_165_axes_0"), val = tensor([-1])]; fp16 var_3107_to_fp16 = const()[name = string("op_3107_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_165_cast_fp16 = layer_norm(axes = normed_165_axes_0, epsilon = var_3107_to_fp16, x = input_173)[name = string("normed_165_cast_fp16")]; tensor var_3117_split_sizes_0 = const()[name = string("op_3117_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3117_axis_0 = const()[name = string("op_3117_axis_0"), val = int32(-1)]; tensor var_3117_0, tensor var_3117_1 = split(axis = var_3117_axis_0, split_sizes = var_3117_split_sizes_0, x = normed_165_cast_fp16)[name = string("op_3117")]; tensor hidden_states_43 = mul(x = var_3117_0, y = layers_5_post_feedforward_layernorm_weight)[name = string("hidden_states_43")]; tensor hidden_states_45_cast_fp16 = add(x = x_113_cast_fp16, y = hidden_states_43)[name = string("hidden_states_45_cast_fp16")]; tensor per_layer_slice_11_begin_0 = const()[name = string("per_layer_slice_11_begin_0"), val = tensor([0, 0, 1280])]; tensor per_layer_slice_11_end_0 = const()[name = string("per_layer_slice_11_end_0"), val = tensor([1, 8, 1536])]; tensor per_layer_slice_11_end_mask_0 = const()[name = string("per_layer_slice_11_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_11_cast_fp16 = slice_by_index(begin = per_layer_slice_11_begin_0, end = per_layer_slice_11_end_0, end_mask = per_layer_slice_11_end_mask_0, x = per_layer_combined_out)[name = string("per_layer_slice_11_cast_fp16")]; tensor gated_21 = linear(bias = linear_2_bias_0, weight = layers_5_per_layer_input_gate_weight_palettized, x = hidden_states_45_cast_fp16)[name = string("linear_53")]; string gated_23_mode_0 = const()[name = string("gated_23_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_23 = gelu(mode = gated_23_mode_0, x = gated_21)[name = string("gated_23")]; tensor input_177_cast_fp16 = mul(x = gated_23, y = per_layer_slice_11_cast_fp16)[name = string("input_177_cast_fp16")]; tensor layers_5_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151712320))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151908992))))[name = string("layers_5_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_54_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_5_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_177_cast_fp16)[name = string("linear_54_cast_fp16")]; int32 var_3155 = const()[name = string("op_3155"), val = int32(-1)]; fp16 const_69_promoted_to_fp16 = const()[name = string("const_69_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3157_cast_fp16 = mul(x = linear_54_cast_fp16, y = const_69_promoted_to_fp16)[name = string("op_3157_cast_fp16")]; bool input_179_interleave_0 = const()[name = string("input_179_interleave_0"), val = bool(false)]; tensor input_179_cast_fp16 = concat(axis = var_3155, interleave = input_179_interleave_0, values = (linear_54_cast_fp16, var_3157_cast_fp16))[name = string("input_179_cast_fp16")]; tensor normed_169_axes_0 = const()[name = string("normed_169_axes_0"), val = tensor([-1])]; fp16 var_3152_to_fp16 = const()[name = string("op_3152_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_169_cast_fp16 = layer_norm(axes = normed_169_axes_0, epsilon = var_3152_to_fp16, x = input_179_cast_fp16)[name = string("normed_169_cast_fp16")]; tensor var_3162_split_sizes_0 = const()[name = string("op_3162_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3162_axis_0 = const()[name = string("op_3162_axis_0"), val = int32(-1)]; tensor var_3162_cast_fp16_0, tensor var_3162_cast_fp16_1 = split(axis = var_3162_axis_0, split_sizes = var_3162_split_sizes_0, x = normed_169_cast_fp16)[name = string("op_3162_cast_fp16")]; tensor layers_5_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_5_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151910592)))]; tensor hidden_states_47_cast_fp16 = mul(x = var_3162_cast_fp16_0, y = layers_5_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_47_cast_fp16")]; tensor hidden_states_49_cast_fp16 = add(x = hidden_states_45_cast_fp16, y = hidden_states_47_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; tensor const_70_promoted_to_fp16 = const()[name = string("const_70_promoted_to_fp16"), val = tensor([0x1.46p-1])]; tensor x_119_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_70_promoted_to_fp16)[name = string("x_119_cast_fp16")]; int32 var_3177 = const()[name = string("op_3177"), val = int32(-1)]; fp16 const_71_promoted_to_fp16 = const()[name = string("const_71_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3179_cast_fp16 = mul(x = x_119_cast_fp16, y = const_71_promoted_to_fp16)[name = string("op_3179_cast_fp16")]; bool input_181_interleave_0 = const()[name = string("input_181_interleave_0"), val = bool(false)]; tensor input_181_cast_fp16 = concat(axis = var_3177, interleave = input_181_interleave_0, values = (x_119_cast_fp16, var_3179_cast_fp16))[name = string("input_181_cast_fp16")]; tensor normed_173_axes_0 = const()[name = string("normed_173_axes_0"), val = tensor([-1])]; fp16 var_3174_to_fp16 = const()[name = string("op_3174_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_173_cast_fp16 = layer_norm(axes = normed_173_axes_0, epsilon = var_3174_to_fp16, x = input_181_cast_fp16)[name = string("normed_173_cast_fp16")]; tensor var_3184_split_sizes_0 = const()[name = string("op_3184_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3184_axis_0 = const()[name = string("op_3184_axis_0"), val = int32(-1)]; tensor var_3184_cast_fp16_0, tensor var_3184_cast_fp16_1 = split(axis = var_3184_axis_0, split_sizes = var_3184_split_sizes_0, x = normed_173_cast_fp16)[name = string("op_3184_cast_fp16")]; tensor layers_6_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_6_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151913728)))]; tensor var_3186_cast_fp16 = mul(x = var_3184_cast_fp16_0, y = layers_6_input_layernorm_weight_promoted_to_fp16)[name = string("op_3186_cast_fp16")]; tensor var_3194 = linear(bias = linear_1_bias_0, weight = layers_6_self_attn_q_proj_weight_palettized, x = var_3186_cast_fp16)[name = string("linear_55")]; tensor var_3199 = const()[name = string("op_3199"), val = tensor([1, 8, 8, 256])]; tensor var_3200 = reshape(shape = var_3199, x = var_3194)[name = string("op_3200")]; tensor var_3205 = const()[name = string("op_3205"), val = tensor([0, 2, 1, 3])]; int32 var_3222 = const()[name = string("op_3222"), val = int32(-1)]; fp16 const_72_promoted = const()[name = string("const_72_promoted"), val = fp16(-0x1p+0)]; tensor var_3206 = transpose(perm = var_3205, x = var_3200)[name = string("transpose_19")]; tensor var_3224 = mul(x = var_3206, y = const_72_promoted)[name = string("op_3224")]; bool input_185_interleave_0 = const()[name = string("input_185_interleave_0"), val = bool(false)]; tensor input_185 = concat(axis = var_3222, interleave = input_185_interleave_0, values = (var_3206, var_3224))[name = string("input_185")]; tensor normed_177_axes_0 = const()[name = string("normed_177_axes_0"), val = tensor([-1])]; fp16 var_3219_to_fp16 = const()[name = string("op_3219_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_177_cast_fp16 = layer_norm(axes = normed_177_axes_0, epsilon = var_3219_to_fp16, x = input_185)[name = string("normed_177_cast_fp16")]; tensor var_3229_split_sizes_0 = const()[name = string("op_3229_split_sizes_0"), val = tensor([256, 256])]; int32 var_3229_axis_0 = const()[name = string("op_3229_axis_0"), val = int32(-1)]; tensor var_3229_0, tensor var_3229_1 = split(axis = var_3229_axis_0, split_sizes = var_3229_split_sizes_0, x = normed_177_cast_fp16)[name = string("op_3229")]; tensor q_51 = mul(x = var_3229_0, y = layers_1_self_attn_q_norm_weight)[name = string("q_51")]; tensor var_3232_cast_fp16 = mul(x = q_51, y = cos_s)[name = string("op_3232_cast_fp16")]; tensor var_3233_split_sizes_0 = const()[name = string("op_3233_split_sizes_0"), val = tensor([128, 128])]; int32 var_3233_axis_0 = const()[name = string("op_3233_axis_0"), val = int32(-1)]; tensor var_3233_0, tensor var_3233_1 = split(axis = var_3233_axis_0, split_sizes = var_3233_split_sizes_0, x = q_51)[name = string("op_3233")]; fp16 const_73_promoted = const()[name = string("const_73_promoted"), val = fp16(-0x1p+0)]; tensor var_3235 = mul(x = var_3233_1, y = const_73_promoted)[name = string("op_3235")]; int32 var_3237 = const()[name = string("op_3237"), val = int32(-1)]; bool var_3238_interleave_0 = const()[name = string("op_3238_interleave_0"), val = bool(false)]; tensor var_3238 = concat(axis = var_3237, interleave = var_3238_interleave_0, values = (var_3235, var_3233_0))[name = string("op_3238")]; tensor var_3239_cast_fp16 = mul(x = var_3238, y = sin_s)[name = string("op_3239_cast_fp16")]; tensor q_55_cast_fp16 = add(x = var_3232_cast_fp16, y = var_3239_cast_fp16)[name = string("q_55_cast_fp16")]; tensor var_3244 = linear(bias = linear_2_bias_0, weight = layers_6_self_attn_k_proj_weight_palettized, x = var_3186_cast_fp16)[name = string("linear_56")]; tensor var_3249 = const()[name = string("op_3249"), val = tensor([1, 8, 1, 256])]; tensor var_3250 = reshape(shape = var_3249, x = var_3244)[name = string("op_3250")]; tensor var_3255 = const()[name = string("op_3255"), val = tensor([0, 2, 1, 3])]; tensor var_3264 = linear(bias = linear_2_bias_0, weight = layers_6_self_attn_v_proj_weight_palettized, x = var_3186_cast_fp16)[name = string("linear_57")]; tensor var_3269 = const()[name = string("op_3269"), val = tensor([1, 8, 1, 256])]; tensor var_3270 = reshape(shape = var_3269, x = var_3264)[name = string("op_3270")]; tensor var_3275 = const()[name = string("op_3275"), val = tensor([0, 2, 1, 3])]; int32 var_3292 = const()[name = string("op_3292"), val = int32(-1)]; fp16 const_74_promoted = const()[name = string("const_74_promoted"), val = fp16(-0x1p+0)]; tensor var_3256 = transpose(perm = var_3255, x = var_3250)[name = string("transpose_18")]; tensor var_3294 = mul(x = var_3256, y = const_74_promoted)[name = string("op_3294")]; bool input_187_interleave_0 = const()[name = string("input_187_interleave_0"), val = bool(false)]; tensor input_187 = concat(axis = var_3292, interleave = input_187_interleave_0, values = (var_3256, var_3294))[name = string("input_187")]; tensor normed_181_axes_0 = const()[name = string("normed_181_axes_0"), val = tensor([-1])]; fp16 var_3289_to_fp16 = const()[name = string("op_3289_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_181_cast_fp16 = layer_norm(axes = normed_181_axes_0, epsilon = var_3289_to_fp16, x = input_187)[name = string("normed_181_cast_fp16")]; tensor var_3299_split_sizes_0 = const()[name = string("op_3299_split_sizes_0"), val = tensor([256, 256])]; int32 var_3299_axis_0 = const()[name = string("op_3299_axis_0"), val = int32(-1)]; tensor var_3299_0, tensor var_3299_1 = split(axis = var_3299_axis_0, split_sizes = var_3299_split_sizes_0, x = normed_181_cast_fp16)[name = string("op_3299")]; tensor q_53 = mul(x = var_3299_0, y = layers_1_self_attn_k_norm_weight)[name = string("q_53")]; fp16 var_3302_promoted = const()[name = string("op_3302_promoted"), val = fp16(0x1p+1)]; tensor var_3276 = transpose(perm = var_3275, x = var_3270)[name = string("transpose_17")]; tensor var_3303 = pow(x = var_3276, y = var_3302_promoted)[name = string("op_3303")]; tensor var_3308_axes_0 = const()[name = string("op_3308_axes_0"), val = tensor([-1])]; bool var_3308_keep_dims_0 = const()[name = string("op_3308_keep_dims_0"), val = bool(true)]; tensor var_3308 = reduce_mean(axes = var_3308_axes_0, keep_dims = var_3308_keep_dims_0, x = var_3303)[name = string("op_3308")]; fp16 var_3310_to_fp16 = const()[name = string("op_3310_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_13_cast_fp16 = add(x = var_3308, y = var_3310_to_fp16)[name = string("mean_sq_13_cast_fp16")]; fp32 var_3312_epsilon_0 = const()[name = string("op_3312_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_3312_cast_fp16 = rsqrt(epsilon = var_3312_epsilon_0, x = mean_sq_13_cast_fp16)[name = string("op_3312_cast_fp16")]; tensor input_191_cast_fp16 = mul(x = var_3276, y = var_3312_cast_fp16)[name = string("input_191_cast_fp16")]; tensor var_3314_cast_fp16 = mul(x = q_53, y = cos_s)[name = string("op_3314_cast_fp16")]; tensor var_3315_split_sizes_0 = const()[name = string("op_3315_split_sizes_0"), val = tensor([128, 128])]; int32 var_3315_axis_0 = const()[name = string("op_3315_axis_0"), val = int32(-1)]; tensor var_3315_0, tensor var_3315_1 = split(axis = var_3315_axis_0, split_sizes = var_3315_split_sizes_0, x = q_53)[name = string("op_3315")]; fp16 const_75_promoted = const()[name = string("const_75_promoted"), val = fp16(-0x1p+0)]; tensor var_3317 = mul(x = var_3315_1, y = const_75_promoted)[name = string("op_3317")]; int32 var_3319 = const()[name = string("op_3319"), val = int32(-1)]; bool var_3320_interleave_0 = const()[name = string("op_3320_interleave_0"), val = bool(false)]; tensor var_3320 = concat(axis = var_3319, interleave = var_3320_interleave_0, values = (var_3317, var_3315_0))[name = string("op_3320")]; tensor var_3321_cast_fp16 = mul(x = var_3320, y = sin_s)[name = string("op_3321_cast_fp16")]; tensor input_189_cast_fp16 = add(x = var_3314_cast_fp16, y = var_3321_cast_fp16)[name = string("input_189_cast_fp16")]; tensor k_padded_11_pad_0 = const()[name = string("k_padded_11_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_11_mode_0 = const()[name = string("k_padded_11_mode_0"), val = string("constant")]; fp16 const_76_to_fp16 = const()[name = string("const_76_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_11_cast_fp16 = pad(constant_val = const_76_to_fp16, mode = k_padded_11_mode_0, pad = k_padded_11_pad_0, x = input_189_cast_fp16)[name = string("k_padded_11_cast_fp16")]; tensor v_padded_11_pad_0 = const()[name = string("v_padded_11_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_11_mode_0 = const()[name = string("v_padded_11_mode_0"), val = string("constant")]; fp16 const_77_to_fp16 = const()[name = string("const_77_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_11_cast_fp16 = pad(constant_val = const_77_to_fp16, mode = v_padded_11_mode_0, pad = v_padded_11_pad_0, x = input_191_cast_fp16)[name = string("v_padded_11_cast_fp16")]; tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([10])]; tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([11])]; int32 concat_74_axis_0 = const()[name = string("concat_74_axis_0"), val = int32(0)]; bool concat_74_interleave_0 = const()[name = string("concat_74_interleave_0"), val = bool(false)]; tensor concat_74 = concat(axis = concat_74_axis_0, interleave = concat_74_interleave_0, values = (expand_dims_72, expand_dims_73, ring_pos, expand_dims_75))[name = string("concat_74")]; tensor concat_75_values1_0 = const()[name = string("concat_75_values1_0"), val = tensor([0])]; tensor concat_75_values3_0 = const()[name = string("concat_75_values3_0"), val = tensor([0])]; int32 concat_75_axis_0 = const()[name = string("concat_75_axis_0"), val = int32(0)]; bool concat_75_interleave_0 = const()[name = string("concat_75_interleave_0"), val = bool(false)]; tensor concat_75 = concat(axis = concat_75_axis_0, interleave = concat_75_interleave_0, values = (expand_dims_76, concat_75_values1_0, var_716, concat_75_values3_0))[name = string("concat_75")]; tensor kv_cache_sliding_internal_tensor_assign_11_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_11_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_11_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_74, begin_mask = kv_cache_sliding_internal_tensor_assign_11_begin_mask_0, end = concat_75, end_mask = kv_cache_sliding_internal_tensor_assign_11_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_11_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_11_stride_0, update = k_padded_11_cast_fp16, x = coreml_update_state_27)[name = string("kv_cache_sliding_internal_tensor_assign_11_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_11_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_28_write_state")]; tensor coreml_update_state_28 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_28")]; tensor expand_dims_78 = const()[name = string("expand_dims_78"), val = tensor([11])]; tensor expand_dims_79 = const()[name = string("expand_dims_79"), val = tensor([0])]; tensor expand_dims_81 = const()[name = string("expand_dims_81"), val = tensor([0])]; tensor expand_dims_82 = const()[name = string("expand_dims_82"), val = tensor([12])]; int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_78, expand_dims_79, ring_pos, expand_dims_81))[name = string("concat_78")]; tensor concat_79_values1_0 = const()[name = string("concat_79_values1_0"), val = tensor([0])]; tensor concat_79_values3_0 = const()[name = string("concat_79_values3_0"), val = tensor([0])]; int32 concat_79_axis_0 = const()[name = string("concat_79_axis_0"), val = int32(0)]; bool concat_79_interleave_0 = const()[name = string("concat_79_interleave_0"), val = bool(false)]; tensor concat_79 = concat(axis = concat_79_axis_0, interleave = concat_79_interleave_0, values = (expand_dims_82, concat_79_values1_0, var_716, concat_79_values3_0))[name = string("concat_79")]; tensor kv_cache_sliding_internal_tensor_assign_12_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_12_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_12_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_78, begin_mask = kv_cache_sliding_internal_tensor_assign_12_begin_mask_0, end = concat_79, end_mask = kv_cache_sliding_internal_tensor_assign_12_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_12_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_12_stride_0, update = v_padded_11_cast_fp16, x = coreml_update_state_28)[name = string("kv_cache_sliding_internal_tensor_assign_12_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_12_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_29_write_state")]; tensor coreml_update_state_29 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_29")]; tensor var_3388_begin_0 = const()[name = string("op_3388_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_3388_end_0 = const()[name = string("op_3388_end_0"), val = tensor([11, 1, 512, 512])]; tensor var_3388_end_mask_0 = const()[name = string("op_3388_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3388_cast_fp16 = slice_by_index(begin = var_3388_begin_0, end = var_3388_end_0, end_mask = var_3388_end_mask_0, x = coreml_update_state_29)[name = string("op_3388_cast_fp16")]; tensor K_sliding_slice_11_begin_0 = const()[name = string("K_sliding_slice_11_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_sliding_slice_11_end_0 = const()[name = string("K_sliding_slice_11_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_sliding_slice_11_end_mask_0 = const()[name = string("K_sliding_slice_11_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_sliding_slice_11_cast_fp16 = slice_by_index(begin = K_sliding_slice_11_begin_0, end = K_sliding_slice_11_end_0, end_mask = K_sliding_slice_11_end_mask_0, x = var_3388_cast_fp16)[name = string("K_sliding_slice_11_cast_fp16")]; tensor var_3408_begin_0 = const()[name = string("op_3408_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_3408_end_0 = const()[name = string("op_3408_end_0"), val = tensor([12, 1, 512, 512])]; tensor var_3408_end_mask_0 = const()[name = string("op_3408_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3408_cast_fp16 = slice_by_index(begin = var_3408_begin_0, end = var_3408_end_0, end_mask = var_3408_end_mask_0, x = coreml_update_state_29)[name = string("op_3408_cast_fp16")]; tensor V_for_attn_11_begin_0 = const()[name = string("V_for_attn_11_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_11_end_0 = const()[name = string("V_for_attn_11_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_11_end_mask_0 = const()[name = string("V_for_attn_11_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_11_cast_fp16 = slice_by_index(begin = V_for_attn_11_begin_0, end = V_for_attn_11_end_0, end_mask = V_for_attn_11_end_mask_0, x = var_3408_cast_fp16)[name = string("V_for_attn_11_cast_fp16")]; tensor transpose_24_perm_0 = const()[name = string("transpose_24_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_12_reps_0 = const()[name = string("tile_12_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_24_cast_fp16 = transpose(perm = transpose_24_perm_0, x = K_sliding_slice_11_cast_fp16)[name = string("transpose_16")]; tensor tile_12_cast_fp16 = tile(reps = tile_12_reps_0, x = transpose_24_cast_fp16)[name = string("tile_12_cast_fp16")]; tensor concat_80 = const()[name = string("concat_80"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_24_cast_fp16 = reshape(shape = concat_80, x = tile_12_cast_fp16)[name = string("reshape_24_cast_fp16")]; tensor transpose_25_perm_0 = const()[name = string("transpose_25_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_81 = const()[name = string("concat_81"), val = tensor([-1, 1, 512, 256])]; tensor transpose_25_cast_fp16 = transpose(perm = transpose_25_perm_0, x = reshape_24_cast_fp16)[name = string("transpose_15")]; tensor reshape_25_cast_fp16 = reshape(shape = concat_81, x = transpose_25_cast_fp16)[name = string("reshape_25_cast_fp16")]; tensor transpose_38_perm_0 = const()[name = string("transpose_38_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_26_perm_0 = const()[name = string("transpose_26_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_13_reps_0 = const()[name = string("tile_13_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_26_cast_fp16 = transpose(perm = transpose_26_perm_0, x = V_for_attn_11_cast_fp16)[name = string("transpose_14")]; tensor tile_13_cast_fp16 = tile(reps = tile_13_reps_0, x = transpose_26_cast_fp16)[name = string("tile_13_cast_fp16")]; tensor concat_82 = const()[name = string("concat_82"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_26_cast_fp16 = reshape(shape = concat_82, x = tile_13_cast_fp16)[name = string("reshape_26_cast_fp16")]; tensor transpose_27_perm_0 = const()[name = string("transpose_27_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_83 = const()[name = string("concat_83"), val = tensor([-1, 1, 512, 256])]; tensor transpose_27_cast_fp16 = transpose(perm = transpose_27_perm_0, x = reshape_26_cast_fp16)[name = string("transpose_13")]; tensor reshape_27_cast_fp16 = reshape(shape = concat_83, x = transpose_27_cast_fp16)[name = string("reshape_27_cast_fp16")]; tensor V_expanded_13_perm_0 = const()[name = string("V_expanded_13_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_25_transpose_x_0 = const()[name = string("attn_weights_25_transpose_x_0"), val = bool(false)]; bool attn_weights_25_transpose_y_0 = const()[name = string("attn_weights_25_transpose_y_0"), val = bool(false)]; tensor transpose_38_cast_fp16 = transpose(perm = transpose_38_perm_0, x = reshape_25_cast_fp16)[name = string("transpose_12")]; tensor attn_weights_25_cast_fp16 = matmul(transpose_x = attn_weights_25_transpose_x_0, transpose_y = attn_weights_25_transpose_y_0, x = q_55_cast_fp16, y = transpose_38_cast_fp16)[name = string("attn_weights_25_cast_fp16")]; tensor x_127_cast_fp16 = add(x = attn_weights_25_cast_fp16, y = causal_mask_sliding)[name = string("x_127_cast_fp16")]; tensor reduce_max_6_axes_0 = const()[name = string("reduce_max_6_axes_0"), val = tensor([-1])]; bool reduce_max_6_keep_dims_0 = const()[name = string("reduce_max_6_keep_dims_0"), val = bool(true)]; tensor reduce_max_6 = reduce_max(axes = reduce_max_6_axes_0, keep_dims = reduce_max_6_keep_dims_0, x = x_127_cast_fp16)[name = string("reduce_max_6")]; tensor var_3453 = sub(x = x_127_cast_fp16, y = reduce_max_6)[name = string("op_3453")]; tensor var_3459 = exp(x = var_3453)[name = string("op_3459")]; tensor var_3469_axes_0 = const()[name = string("op_3469_axes_0"), val = tensor([-1])]; bool var_3469_keep_dims_0 = const()[name = string("op_3469_keep_dims_0"), val = bool(true)]; tensor var_3469 = reduce_sum(axes = var_3469_axes_0, keep_dims = var_3469_keep_dims_0, x = var_3459)[name = string("op_3469")]; tensor var_3475_cast_fp16 = real_div(x = var_3459, y = var_3469)[name = string("op_3475_cast_fp16")]; bool attn_output_25_transpose_x_0 = const()[name = string("attn_output_25_transpose_x_0"), val = bool(false)]; bool attn_output_25_transpose_y_0 = const()[name = string("attn_output_25_transpose_y_0"), val = bool(false)]; tensor V_expanded_13_cast_fp16 = transpose(perm = V_expanded_13_perm_0, x = reshape_27_cast_fp16)[name = string("transpose_11")]; tensor attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_0, transpose_y = attn_output_25_transpose_y_0, x = var_3475_cast_fp16, y = V_expanded_13_cast_fp16)[name = string("attn_output_25_cast_fp16")]; tensor var_3486 = const()[name = string("op_3486"), val = tensor([0, 2, 1, 3])]; tensor var_3493 = const()[name = string("op_3493"), val = tensor([1, 8, -1])]; tensor var_3487_cast_fp16 = transpose(perm = var_3486, x = attn_output_25_cast_fp16)[name = string("transpose_10")]; tensor input_193_cast_fp16 = reshape(shape = var_3493, x = var_3487_cast_fp16)[name = string("input_193_cast_fp16")]; tensor layers_6_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(151916864))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153489792))))[name = string("layers_6_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_58_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_6_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_193_cast_fp16)[name = string("linear_58_cast_fp16")]; int32 var_3502 = const()[name = string("op_3502"), val = int32(-1)]; fp16 const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3504_cast_fp16 = mul(x = linear_58_cast_fp16, y = const_78_promoted_to_fp16)[name = string("op_3504_cast_fp16")]; bool input_195_interleave_0 = const()[name = string("input_195_interleave_0"), val = bool(false)]; tensor input_195_cast_fp16 = concat(axis = var_3502, interleave = input_195_interleave_0, values = (linear_58_cast_fp16, var_3504_cast_fp16))[name = string("input_195_cast_fp16")]; tensor normed_185_axes_0 = const()[name = string("normed_185_axes_0"), val = tensor([-1])]; fp16 var_3499_to_fp16 = const()[name = string("op_3499_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_185_cast_fp16 = layer_norm(axes = normed_185_axes_0, epsilon = var_3499_to_fp16, x = input_195_cast_fp16)[name = string("normed_185_cast_fp16")]; tensor var_3509_split_sizes_0 = const()[name = string("op_3509_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3509_axis_0 = const()[name = string("op_3509_axis_0"), val = int32(-1)]; tensor var_3509_cast_fp16_0, tensor var_3509_cast_fp16_1 = split(axis = var_3509_axis_0, split_sizes = var_3509_split_sizes_0, x = normed_185_cast_fp16)[name = string("op_3509_cast_fp16")]; tensor layers_6_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_6_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153491392)))]; tensor attn_output_27_cast_fp16 = mul(x = var_3509_cast_fp16_0, y = layers_6_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_27_cast_fp16")]; tensor x_133_cast_fp16 = add(x = x_119_cast_fp16, y = attn_output_27_cast_fp16)[name = string("x_133_cast_fp16")]; int32 var_3518 = const()[name = string("op_3518"), val = int32(-1)]; fp16 const_79_promoted_to_fp16 = const()[name = string("const_79_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3520_cast_fp16 = mul(x = x_133_cast_fp16, y = const_79_promoted_to_fp16)[name = string("op_3520_cast_fp16")]; bool input_197_interleave_0 = const()[name = string("input_197_interleave_0"), val = bool(false)]; tensor input_197_cast_fp16 = concat(axis = var_3518, interleave = input_197_interleave_0, values = (x_133_cast_fp16, var_3520_cast_fp16))[name = string("input_197_cast_fp16")]; tensor normed_189_axes_0 = const()[name = string("normed_189_axes_0"), val = tensor([-1])]; fp16 var_3515_to_fp16 = const()[name = string("op_3515_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_189_cast_fp16 = layer_norm(axes = normed_189_axes_0, epsilon = var_3515_to_fp16, x = input_197_cast_fp16)[name = string("normed_189_cast_fp16")]; tensor var_3525_split_sizes_0 = const()[name = string("op_3525_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3525_axis_0 = const()[name = string("op_3525_axis_0"), val = int32(-1)]; tensor var_3525_cast_fp16_0, tensor var_3525_cast_fp16_1 = split(axis = var_3525_axis_0, split_sizes = var_3525_split_sizes_0, x = normed_189_cast_fp16)[name = string("op_3525_cast_fp16")]; tensor layers_6_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_6_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153494528)))]; tensor var_3527_cast_fp16 = mul(x = var_3525_cast_fp16_0, y = layers_6_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("op_3527_cast_fp16")]; tensor gate_25 = linear(bias = linear_5_bias_0, weight = layers_6_mlp_gate_proj_weight_palettized, x = var_3527_cast_fp16)[name = string("linear_59")]; tensor up_13 = linear(bias = linear_5_bias_0, weight = layers_6_mlp_up_proj_weight_palettized, x = var_3527_cast_fp16)[name = string("linear_60")]; string gate_27_mode_0 = const()[name = string("gate_27_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate_27 = gelu(mode = gate_27_mode_0, x = gate_25)[name = string("gate_27")]; tensor input_201 = mul(x = gate_27, y = up_13)[name = string("input_201")]; tensor x_135 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_6_mlp_down_proj_weight_palettized, x = input_201)[name = string("linear_61")]; int32 var_3549 = const()[name = string("op_3549"), val = int32(-1)]; fp16 const_80_promoted = const()[name = string("const_80_promoted"), val = fp16(-0x1p+0)]; tensor var_3551 = mul(x = x_135, y = const_80_promoted)[name = string("op_3551")]; bool input_203_interleave_0 = const()[name = string("input_203_interleave_0"), val = bool(false)]; tensor input_203 = concat(axis = var_3549, interleave = input_203_interleave_0, values = (x_135, var_3551))[name = string("input_203")]; tensor normed_193_axes_0 = const()[name = string("normed_193_axes_0"), val = tensor([-1])]; fp16 var_3546_to_fp16 = const()[name = string("op_3546_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_193_cast_fp16 = layer_norm(axes = normed_193_axes_0, epsilon = var_3546_to_fp16, x = input_203)[name = string("normed_193_cast_fp16")]; tensor var_3556_split_sizes_0 = const()[name = string("op_3556_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3556_axis_0 = const()[name = string("op_3556_axis_0"), val = int32(-1)]; tensor var_3556_0, tensor var_3556_1 = split(axis = var_3556_axis_0, split_sizes = var_3556_split_sizes_0, x = normed_193_cast_fp16)[name = string("op_3556")]; tensor hidden_states_51 = mul(x = var_3556_0, y = layers_6_post_feedforward_layernorm_weight)[name = string("hidden_states_51")]; tensor hidden_states_53_cast_fp16 = add(x = x_133_cast_fp16, y = hidden_states_51)[name = string("hidden_states_53_cast_fp16")]; tensor per_layer_slice_13_begin_0 = const()[name = string("per_layer_slice_13_begin_0"), val = tensor([0, 0, 1536])]; tensor per_layer_slice_13_end_0 = const()[name = string("per_layer_slice_13_end_0"), val = tensor([1, 8, 1792])]; tensor per_layer_slice_13_end_mask_0 = const()[name = string("per_layer_slice_13_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_13_cast_fp16 = slice_by_index(begin = per_layer_slice_13_begin_0, end = per_layer_slice_13_end_0, end_mask = per_layer_slice_13_end_mask_0, x = per_layer_combined_out)[name = string("per_layer_slice_13_cast_fp16")]; tensor gated_25 = linear(bias = linear_2_bias_0, weight = layers_6_per_layer_input_gate_weight_palettized, x = hidden_states_53_cast_fp16)[name = string("linear_62")]; string gated_27_mode_0 = const()[name = string("gated_27_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated_27 = gelu(mode = gated_27_mode_0, x = gated_25)[name = string("gated_27")]; tensor input_207_cast_fp16 = mul(x = gated_27, y = per_layer_slice_13_cast_fp16)[name = string("input_207_cast_fp16")]; tensor layers_6_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153497664))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153694336))))[name = string("layers_6_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_63_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_6_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_207_cast_fp16)[name = string("linear_63_cast_fp16")]; int32 var_3594 = const()[name = string("op_3594"), val = int32(-1)]; fp16 const_81_promoted_to_fp16 = const()[name = string("const_81_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3596_cast_fp16 = mul(x = linear_63_cast_fp16, y = const_81_promoted_to_fp16)[name = string("op_3596_cast_fp16")]; bool input_209_interleave_0 = const()[name = string("input_209_interleave_0"), val = bool(false)]; tensor input_209_cast_fp16 = concat(axis = var_3594, interleave = input_209_interleave_0, values = (linear_63_cast_fp16, var_3596_cast_fp16))[name = string("input_209_cast_fp16")]; tensor normed_197_axes_0 = const()[name = string("normed_197_axes_0"), val = tensor([-1])]; fp16 var_3591_to_fp16 = const()[name = string("op_3591_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_197_cast_fp16 = layer_norm(axes = normed_197_axes_0, epsilon = var_3591_to_fp16, x = input_209_cast_fp16)[name = string("normed_197_cast_fp16")]; tensor var_3601_split_sizes_0 = const()[name = string("op_3601_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3601_axis_0 = const()[name = string("op_3601_axis_0"), val = int32(-1)]; tensor var_3601_cast_fp16_0, tensor var_3601_cast_fp16_1 = split(axis = var_3601_axis_0, split_sizes = var_3601_split_sizes_0, x = normed_197_cast_fp16)[name = string("op_3601_cast_fp16")]; tensor layers_6_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_6_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153695936)))]; tensor hidden_states_55_cast_fp16 = mul(x = var_3601_cast_fp16_0, y = layers_6_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_55_cast_fp16")]; tensor hidden_states_57_cast_fp16 = add(x = hidden_states_53_cast_fp16, y = hidden_states_55_cast_fp16)[name = string("hidden_states_57_cast_fp16")]; tensor const_82_promoted_to_fp16 = const()[name = string("const_82_promoted_to_fp16"), val = tensor([0x1.fep-2])]; tensor x_139_cast_fp16 = mul(x = hidden_states_57_cast_fp16, y = const_82_promoted_to_fp16)[name = string("x_139_cast_fp16")]; int32 var_3616 = const()[name = string("op_3616"), val = int32(-1)]; fp16 const_83_promoted_to_fp16 = const()[name = string("const_83_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3618_cast_fp16 = mul(x = x_139_cast_fp16, y = const_83_promoted_to_fp16)[name = string("op_3618_cast_fp16")]; bool input_211_interleave_0 = const()[name = string("input_211_interleave_0"), val = bool(false)]; tensor input_211_cast_fp16 = concat(axis = var_3616, interleave = input_211_interleave_0, values = (x_139_cast_fp16, var_3618_cast_fp16))[name = string("input_211_cast_fp16")]; tensor normed_201_axes_0 = const()[name = string("normed_201_axes_0"), val = tensor([-1])]; fp16 var_3613_to_fp16 = const()[name = string("op_3613_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_201_cast_fp16 = layer_norm(axes = normed_201_axes_0, epsilon = var_3613_to_fp16, x = input_211_cast_fp16)[name = string("normed_201_cast_fp16")]; tensor var_3623_split_sizes_0 = const()[name = string("op_3623_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3623_axis_0 = const()[name = string("op_3623_axis_0"), val = int32(-1)]; tensor var_3623_cast_fp16_0, tensor var_3623_cast_fp16_1 = split(axis = var_3623_axis_0, split_sizes = var_3623_split_sizes_0, x = normed_201_cast_fp16)[name = string("op_3623_cast_fp16")]; tensor layers_7_input_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_7_input_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153699072)))]; tensor var_3625_cast_fp16 = mul(x = var_3623_cast_fp16_0, y = layers_7_input_layernorm_weight_promoted_to_fp16)[name = string("op_3625_cast_fp16")]; tensor var_3633 = linear(bias = linear_1_bias_0, weight = layers_7_self_attn_q_proj_weight_palettized, x = var_3625_cast_fp16)[name = string("linear_64")]; tensor var_3638 = const()[name = string("op_3638"), val = tensor([1, 8, 8, 256])]; tensor var_3639 = reshape(shape = var_3638, x = var_3633)[name = string("op_3639")]; tensor var_3644 = const()[name = string("op_3644"), val = tensor([0, 2, 1, 3])]; int32 var_3661 = const()[name = string("op_3661"), val = int32(-1)]; fp16 const_84_promoted = const()[name = string("const_84_promoted"), val = fp16(-0x1p+0)]; tensor var_3645 = transpose(perm = var_3644, x = var_3639)[name = string("transpose_9")]; tensor var_3663 = mul(x = var_3645, y = const_84_promoted)[name = string("op_3663")]; bool input_215_interleave_0 = const()[name = string("input_215_interleave_0"), val = bool(false)]; tensor input_215 = concat(axis = var_3661, interleave = input_215_interleave_0, values = (var_3645, var_3663))[name = string("input_215")]; tensor normed_205_axes_0 = const()[name = string("normed_205_axes_0"), val = tensor([-1])]; fp16 var_3658_to_fp16 = const()[name = string("op_3658_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_205_cast_fp16 = layer_norm(axes = normed_205_axes_0, epsilon = var_3658_to_fp16, x = input_215)[name = string("normed_205_cast_fp16")]; tensor var_3668_split_sizes_0 = const()[name = string("op_3668_split_sizes_0"), val = tensor([256, 256])]; int32 var_3668_axis_0 = const()[name = string("op_3668_axis_0"), val = int32(-1)]; tensor var_3668_0, tensor var_3668_1 = split(axis = var_3668_axis_0, split_sizes = var_3668_split_sizes_0, x = normed_205_cast_fp16)[name = string("op_3668")]; tensor q_59 = mul(x = var_3668_0, y = layers_7_self_attn_q_norm_weight)[name = string("q_59")]; tensor var_3671_cast_fp16 = mul(x = q_59, y = cos_s)[name = string("op_3671_cast_fp16")]; tensor var_3672_split_sizes_0 = const()[name = string("op_3672_split_sizes_0"), val = tensor([128, 128])]; int32 var_3672_axis_0 = const()[name = string("op_3672_axis_0"), val = int32(-1)]; tensor var_3672_0, tensor var_3672_1 = split(axis = var_3672_axis_0, split_sizes = var_3672_split_sizes_0, x = q_59)[name = string("op_3672")]; fp16 const_85_promoted = const()[name = string("const_85_promoted"), val = fp16(-0x1p+0)]; tensor var_3674 = mul(x = var_3672_1, y = const_85_promoted)[name = string("op_3674")]; int32 var_3676 = const()[name = string("op_3676"), val = int32(-1)]; bool var_3677_interleave_0 = const()[name = string("op_3677_interleave_0"), val = bool(false)]; tensor var_3677 = concat(axis = var_3676, interleave = var_3677_interleave_0, values = (var_3674, var_3672_0))[name = string("op_3677")]; tensor var_3678_cast_fp16 = mul(x = var_3677, y = sin_s)[name = string("op_3678_cast_fp16")]; tensor q_cast_fp16 = add(x = var_3671_cast_fp16, y = var_3678_cast_fp16)[name = string("q_cast_fp16")]; tensor var_3683 = linear(bias = linear_2_bias_0, weight = layers_7_self_attn_k_proj_weight_palettized, x = var_3625_cast_fp16)[name = string("linear_65")]; tensor var_3688 = const()[name = string("op_3688"), val = tensor([1, 8, 1, 256])]; tensor var_3689 = reshape(shape = var_3688, x = var_3683)[name = string("op_3689")]; tensor var_3694 = const()[name = string("op_3694"), val = tensor([0, 2, 1, 3])]; tensor var_3703 = linear(bias = linear_2_bias_0, weight = layers_7_self_attn_v_proj_weight_palettized, x = var_3625_cast_fp16)[name = string("linear_66")]; tensor var_3708 = const()[name = string("op_3708"), val = tensor([1, 8, 1, 256])]; tensor var_3709 = reshape(shape = var_3708, x = var_3703)[name = string("op_3709")]; tensor var_3714 = const()[name = string("op_3714"), val = tensor([0, 2, 1, 3])]; int32 var_3731 = const()[name = string("op_3731"), val = int32(-1)]; fp16 const_86_promoted = const()[name = string("const_86_promoted"), val = fp16(-0x1p+0)]; tensor var_3695 = transpose(perm = var_3694, x = var_3689)[name = string("transpose_8")]; tensor var_3733 = mul(x = var_3695, y = const_86_promoted)[name = string("op_3733")]; bool input_217_interleave_0 = const()[name = string("input_217_interleave_0"), val = bool(false)]; tensor input_217 = concat(axis = var_3731, interleave = input_217_interleave_0, values = (var_3695, var_3733))[name = string("input_217")]; tensor normed_209_axes_0 = const()[name = string("normed_209_axes_0"), val = tensor([-1])]; fp16 var_3728_to_fp16 = const()[name = string("op_3728_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_209_cast_fp16 = layer_norm(axes = normed_209_axes_0, epsilon = var_3728_to_fp16, x = input_217)[name = string("normed_209_cast_fp16")]; tensor var_3738_split_sizes_0 = const()[name = string("op_3738_split_sizes_0"), val = tensor([256, 256])]; int32 var_3738_axis_0 = const()[name = string("op_3738_axis_0"), val = int32(-1)]; tensor var_3738_0, tensor var_3738_1 = split(axis = var_3738_axis_0, split_sizes = var_3738_split_sizes_0, x = normed_209_cast_fp16)[name = string("op_3738")]; tensor q_61 = mul(x = var_3738_0, y = layers_7_self_attn_k_norm_weight)[name = string("q_61")]; fp16 var_3741_promoted = const()[name = string("op_3741_promoted"), val = fp16(0x1p+1)]; tensor var_3715 = transpose(perm = var_3714, x = var_3709)[name = string("transpose_7")]; tensor var_3742 = pow(x = var_3715, y = var_3741_promoted)[name = string("op_3742")]; tensor var_3747_axes_0 = const()[name = string("op_3747_axes_0"), val = tensor([-1])]; bool var_3747_keep_dims_0 = const()[name = string("op_3747_keep_dims_0"), val = bool(true)]; tensor var_3747 = reduce_mean(axes = var_3747_axes_0, keep_dims = var_3747_keep_dims_0, x = var_3742)[name = string("op_3747")]; fp16 var_3749_to_fp16 = const()[name = string("op_3749_to_fp16"), val = fp16(0x1.1p-20)]; tensor mean_sq_cast_fp16 = add(x = var_3747, y = var_3749_to_fp16)[name = string("mean_sq_cast_fp16")]; fp32 var_3751_epsilon_0 = const()[name = string("op_3751_epsilon_0"), val = fp32(0x1.197998p-40)]; tensor var_3751_cast_fp16 = rsqrt(epsilon = var_3751_epsilon_0, x = mean_sq_cast_fp16)[name = string("op_3751_cast_fp16")]; tensor input_221_cast_fp16 = mul(x = var_3715, y = var_3751_cast_fp16)[name = string("input_221_cast_fp16")]; tensor var_3753_cast_fp16 = mul(x = q_61, y = cos_s)[name = string("op_3753_cast_fp16")]; tensor var_3754_split_sizes_0 = const()[name = string("op_3754_split_sizes_0"), val = tensor([128, 128])]; int32 var_3754_axis_0 = const()[name = string("op_3754_axis_0"), val = int32(-1)]; tensor var_3754_0, tensor var_3754_1 = split(axis = var_3754_axis_0, split_sizes = var_3754_split_sizes_0, x = q_61)[name = string("op_3754")]; fp16 const_87_promoted = const()[name = string("const_87_promoted"), val = fp16(-0x1p+0)]; tensor var_3756 = mul(x = var_3754_1, y = const_87_promoted)[name = string("op_3756")]; int32 var_3758 = const()[name = string("op_3758"), val = int32(-1)]; bool var_3759_interleave_0 = const()[name = string("op_3759_interleave_0"), val = bool(false)]; tensor var_3759 = concat(axis = var_3758, interleave = var_3759_interleave_0, values = (var_3756, var_3754_0))[name = string("op_3759")]; tensor var_3760_cast_fp16 = mul(x = var_3759, y = sin_s)[name = string("op_3760_cast_fp16")]; tensor input_219_cast_fp16 = add(x = var_3753_cast_fp16, y = var_3760_cast_fp16)[name = string("input_219_cast_fp16")]; tensor k_padded_pad_0 = const()[name = string("k_padded_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string k_padded_mode_0 = const()[name = string("k_padded_mode_0"), val = string("constant")]; fp16 const_88_to_fp16 = const()[name = string("const_88_to_fp16"), val = fp16(0x0p+0)]; tensor k_padded_cast_fp16 = pad(constant_val = const_88_to_fp16, mode = k_padded_mode_0, pad = k_padded_pad_0, x = input_219_cast_fp16)[name = string("k_padded_cast_fp16")]; tensor v_padded_pad_0 = const()[name = string("v_padded_pad_0"), val = tensor([0, 0, 0, 0, 0, 0, 0, 256])]; string v_padded_mode_0 = const()[name = string("v_padded_mode_0"), val = string("constant")]; fp16 const_89_to_fp16 = const()[name = string("const_89_to_fp16"), val = fp16(0x0p+0)]; tensor v_padded_cast_fp16 = pad(constant_val = const_89_to_fp16, mode = v_padded_mode_0, pad = v_padded_pad_0, x = input_221_cast_fp16)[name = string("v_padded_cast_fp16")]; tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([12])]; tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([13])]; int32 concat_86_axis_0 = const()[name = string("concat_86_axis_0"), val = int32(0)]; bool concat_86_interleave_0 = const()[name = string("concat_86_interleave_0"), val = bool(false)]; tensor concat_86 = concat(axis = concat_86_axis_0, interleave = concat_86_interleave_0, values = (expand_dims_84, expand_dims_85, ring_pos, expand_dims_87))[name = string("concat_86")]; tensor concat_87_values1_0 = const()[name = string("concat_87_values1_0"), val = tensor([0])]; tensor concat_87_values3_0 = const()[name = string("concat_87_values3_0"), val = tensor([0])]; int32 concat_87_axis_0 = const()[name = string("concat_87_axis_0"), val = int32(0)]; bool concat_87_interleave_0 = const()[name = string("concat_87_interleave_0"), val = bool(false)]; tensor concat_87 = concat(axis = concat_87_axis_0, interleave = concat_87_interleave_0, values = (expand_dims_88, concat_87_values1_0, var_716, concat_87_values3_0))[name = string("concat_87")]; tensor kv_cache_sliding_internal_tensor_assign_13_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_13_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_13_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_86, begin_mask = kv_cache_sliding_internal_tensor_assign_13_begin_mask_0, end = concat_87, end_mask = kv_cache_sliding_internal_tensor_assign_13_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_13_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_13_stride_0, update = k_padded_cast_fp16, x = coreml_update_state_29)[name = string("kv_cache_sliding_internal_tensor_assign_13_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_13_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_30_write_state")]; tensor coreml_update_state_30 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_30")]; tensor expand_dims_90 = const()[name = string("expand_dims_90"), val = tensor([13])]; tensor expand_dims_91 = const()[name = string("expand_dims_91"), val = tensor([0])]; tensor expand_dims_93 = const()[name = string("expand_dims_93"), val = tensor([0])]; tensor expand_dims_94 = const()[name = string("expand_dims_94"), val = tensor([14])]; int32 concat_90_axis_0 = const()[name = string("concat_90_axis_0"), val = int32(0)]; bool concat_90_interleave_0 = const()[name = string("concat_90_interleave_0"), val = bool(false)]; tensor concat_90 = concat(axis = concat_90_axis_0, interleave = concat_90_interleave_0, values = (expand_dims_90, expand_dims_91, ring_pos, expand_dims_93))[name = string("concat_90")]; tensor concat_91_values1_0 = const()[name = string("concat_91_values1_0"), val = tensor([0])]; tensor concat_91_values3_0 = const()[name = string("concat_91_values3_0"), val = tensor([0])]; int32 concat_91_axis_0 = const()[name = string("concat_91_axis_0"), val = int32(0)]; bool concat_91_interleave_0 = const()[name = string("concat_91_interleave_0"), val = bool(false)]; tensor concat_91 = concat(axis = concat_91_axis_0, interleave = concat_91_interleave_0, values = (expand_dims_94, concat_91_values1_0, var_716, concat_91_values3_0))[name = string("concat_91")]; tensor kv_cache_sliding_internal_tensor_assign_14_stride_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor kv_cache_sliding_internal_tensor_assign_14_begin_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_14_end_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor kv_cache_sliding_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("kv_cache_sliding_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor kv_cache_sliding_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_90, begin_mask = kv_cache_sliding_internal_tensor_assign_14_begin_mask_0, end = concat_91, end_mask = kv_cache_sliding_internal_tensor_assign_14_end_mask_0, squeeze_mask = kv_cache_sliding_internal_tensor_assign_14_squeeze_mask_0, stride = kv_cache_sliding_internal_tensor_assign_14_stride_0, update = v_padded_cast_fp16, x = coreml_update_state_30)[name = string("kv_cache_sliding_internal_tensor_assign_14_cast_fp16")]; write_state(data = kv_cache_sliding_internal_tensor_assign_14_cast_fp16, input = kv_cache_sliding)[name = string("coreml_update_state_31_write_state")]; tensor coreml_update_state_31 = read_state(input = kv_cache_sliding)[name = string("coreml_update_state_31")]; tensor var_3827_begin_0 = const()[name = string("op_3827_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_3827_end_0 = const()[name = string("op_3827_end_0"), val = tensor([13, 1, 512, 512])]; tensor var_3827_end_mask_0 = const()[name = string("op_3827_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3827_cast_fp16 = slice_by_index(begin = var_3827_begin_0, end = var_3827_end_0, end_mask = var_3827_end_mask_0, x = coreml_update_state_31)[name = string("op_3827_cast_fp16")]; tensor K_sliding_slice_begin_0 = const()[name = string("K_sliding_slice_begin_0"), val = tensor([0, 0, 0, 0])]; tensor K_sliding_slice_end_0 = const()[name = string("K_sliding_slice_end_0"), val = tensor([1, 1, 512, 256])]; tensor K_sliding_slice_end_mask_0 = const()[name = string("K_sliding_slice_end_mask_0"), val = tensor([true, true, true, false])]; tensor K_sliding_slice_cast_fp16 = slice_by_index(begin = K_sliding_slice_begin_0, end = K_sliding_slice_end_0, end_mask = K_sliding_slice_end_mask_0, x = var_3827_cast_fp16)[name = string("K_sliding_slice_cast_fp16")]; tensor var_3847_begin_0 = const()[name = string("op_3847_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_3847_end_0 = const()[name = string("op_3847_end_0"), val = tensor([1, 1, 512, 512])]; tensor var_3847_end_mask_0 = const()[name = string("op_3847_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_3847_cast_fp16 = slice_by_index(begin = var_3847_begin_0, end = var_3847_end_0, end_mask = var_3847_end_mask_0, x = coreml_update_state_31)[name = string("op_3847_cast_fp16")]; tensor V_for_attn_begin_0 = const()[name = string("V_for_attn_begin_0"), val = tensor([0, 0, 0, 0])]; tensor V_for_attn_end_0 = const()[name = string("V_for_attn_end_0"), val = tensor([1, 1, 512, 256])]; tensor V_for_attn_end_mask_0 = const()[name = string("V_for_attn_end_mask_0"), val = tensor([true, true, true, false])]; tensor V_for_attn_cast_fp16 = slice_by_index(begin = V_for_attn_begin_0, end = V_for_attn_end_0, end_mask = V_for_attn_end_mask_0, x = var_3847_cast_fp16)[name = string("V_for_attn_cast_fp16")]; tensor transpose_28_perm_0 = const()[name = string("transpose_28_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_14_reps_0 = const()[name = string("tile_14_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_28_cast_fp16 = transpose(perm = transpose_28_perm_0, x = K_sliding_slice_cast_fp16)[name = string("transpose_6")]; tensor tile_14_cast_fp16 = tile(reps = tile_14_reps_0, x = transpose_28_cast_fp16)[name = string("tile_14_cast_fp16")]; tensor concat_92 = const()[name = string("concat_92"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_28_cast_fp16 = reshape(shape = concat_92, x = tile_14_cast_fp16)[name = string("reshape_28_cast_fp16")]; tensor transpose_29_perm_0 = const()[name = string("transpose_29_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_93 = const()[name = string("concat_93"), val = tensor([-1, 1, 512, 256])]; tensor transpose_29_cast_fp16 = transpose(perm = transpose_29_perm_0, x = reshape_28_cast_fp16)[name = string("transpose_5")]; tensor reshape_29_cast_fp16 = reshape(shape = concat_93, x = transpose_29_cast_fp16)[name = string("reshape_29_cast_fp16")]; tensor transpose_39_perm_0 = const()[name = string("transpose_39_perm_0"), val = tensor([1, 0, -1, -2])]; tensor transpose_30_perm_0 = const()[name = string("transpose_30_perm_0"), val = tensor([1, 0, 2, 3])]; tensor tile_15_reps_0 = const()[name = string("tile_15_reps_0"), val = tensor([8, 1, 1, 1])]; tensor transpose_30_cast_fp16 = transpose(perm = transpose_30_perm_0, x = V_for_attn_cast_fp16)[name = string("transpose_4")]; tensor tile_15_cast_fp16 = tile(reps = tile_15_reps_0, x = transpose_30_cast_fp16)[name = string("tile_15_cast_fp16")]; tensor concat_94 = const()[name = string("concat_94"), val = tensor([8, 1, 1, 512, 256])]; tensor reshape_30_cast_fp16 = reshape(shape = concat_94, x = tile_15_cast_fp16)[name = string("reshape_30_cast_fp16")]; tensor transpose_31_perm_0 = const()[name = string("transpose_31_perm_0"), val = tensor([1, 0, 2, 3, 4])]; tensor concat_95 = const()[name = string("concat_95"), val = tensor([-1, 1, 512, 256])]; tensor transpose_31_cast_fp16 = transpose(perm = transpose_31_perm_0, x = reshape_30_cast_fp16)[name = string("transpose_3")]; tensor reshape_31_cast_fp16 = reshape(shape = concat_95, x = transpose_31_cast_fp16)[name = string("reshape_31_cast_fp16")]; tensor V_expanded_perm_0 = const()[name = string("V_expanded_perm_0"), val = tensor([1, 0, -2, -1])]; bool attn_weights_29_transpose_x_0 = const()[name = string("attn_weights_29_transpose_x_0"), val = bool(false)]; bool attn_weights_29_transpose_y_0 = const()[name = string("attn_weights_29_transpose_y_0"), val = bool(false)]; tensor transpose_39_cast_fp16 = transpose(perm = transpose_39_perm_0, x = reshape_29_cast_fp16)[name = string("transpose_2")]; tensor attn_weights_29_cast_fp16 = matmul(transpose_x = attn_weights_29_transpose_x_0, transpose_y = attn_weights_29_transpose_y_0, x = q_cast_fp16, y = transpose_39_cast_fp16)[name = string("attn_weights_29_cast_fp16")]; tensor x_147_cast_fp16 = add(x = attn_weights_29_cast_fp16, y = causal_mask_sliding)[name = string("x_147_cast_fp16")]; tensor reduce_max_7_axes_0 = const()[name = string("reduce_max_7_axes_0"), val = tensor([-1])]; bool reduce_max_7_keep_dims_0 = const()[name = string("reduce_max_7_keep_dims_0"), val = bool(true)]; tensor reduce_max_7 = reduce_max(axes = reduce_max_7_axes_0, keep_dims = reduce_max_7_keep_dims_0, x = x_147_cast_fp16)[name = string("reduce_max_7")]; tensor var_3892 = sub(x = x_147_cast_fp16, y = reduce_max_7)[name = string("op_3892")]; tensor var_3898 = exp(x = var_3892)[name = string("op_3898")]; tensor var_3908_axes_0 = const()[name = string("op_3908_axes_0"), val = tensor([-1])]; bool var_3908_keep_dims_0 = const()[name = string("op_3908_keep_dims_0"), val = bool(true)]; tensor var_3908 = reduce_sum(axes = var_3908_axes_0, keep_dims = var_3908_keep_dims_0, x = var_3898)[name = string("op_3908")]; tensor var_3914_cast_fp16 = real_div(x = var_3898, y = var_3908)[name = string("op_3914_cast_fp16")]; bool attn_output_29_transpose_x_0 = const()[name = string("attn_output_29_transpose_x_0"), val = bool(false)]; bool attn_output_29_transpose_y_0 = const()[name = string("attn_output_29_transpose_y_0"), val = bool(false)]; tensor V_expanded_cast_fp16 = transpose(perm = V_expanded_perm_0, x = reshape_31_cast_fp16)[name = string("transpose_1")]; tensor attn_output_29_cast_fp16 = matmul(transpose_x = attn_output_29_transpose_x_0, transpose_y = attn_output_29_transpose_y_0, x = var_3914_cast_fp16, y = V_expanded_cast_fp16)[name = string("attn_output_29_cast_fp16")]; tensor var_3925 = const()[name = string("op_3925"), val = tensor([0, 2, 1, 3])]; tensor var_3932 = const()[name = string("op_3932"), val = tensor([1, 8, -1])]; tensor var_3926_cast_fp16 = transpose(perm = var_3925, x = attn_output_29_cast_fp16)[name = string("transpose_0")]; tensor input_223_cast_fp16 = reshape(shape = var_3932, x = var_3926_cast_fp16)[name = string("input_223_cast_fp16")]; tensor layers_7_self_attn_o_proj_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(153702208))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(155275136))))[name = string("layers_7_self_attn_o_proj_weight_promoted_to_fp16_palettized")]; tensor linear_67_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_7_self_attn_o_proj_weight_promoted_to_fp16_palettized, x = input_223_cast_fp16)[name = string("linear_67_cast_fp16")]; int32 var_3941 = const()[name = string("op_3941"), val = int32(-1)]; fp16 const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3943_cast_fp16 = mul(x = linear_67_cast_fp16, y = const_90_promoted_to_fp16)[name = string("op_3943_cast_fp16")]; bool input_225_interleave_0 = const()[name = string("input_225_interleave_0"), val = bool(false)]; tensor input_225_cast_fp16 = concat(axis = var_3941, interleave = input_225_interleave_0, values = (linear_67_cast_fp16, var_3943_cast_fp16))[name = string("input_225_cast_fp16")]; tensor normed_213_axes_0 = const()[name = string("normed_213_axes_0"), val = tensor([-1])]; fp16 var_3938_to_fp16 = const()[name = string("op_3938_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_213_cast_fp16 = layer_norm(axes = normed_213_axes_0, epsilon = var_3938_to_fp16, x = input_225_cast_fp16)[name = string("normed_213_cast_fp16")]; tensor var_3948_split_sizes_0 = const()[name = string("op_3948_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3948_axis_0 = const()[name = string("op_3948_axis_0"), val = int32(-1)]; tensor var_3948_cast_fp16_0, tensor var_3948_cast_fp16_1 = split(axis = var_3948_axis_0, split_sizes = var_3948_split_sizes_0, x = normed_213_cast_fp16)[name = string("op_3948_cast_fp16")]; tensor layers_7_post_attention_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_7_post_attention_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(155276736)))]; tensor attn_output_cast_fp16 = mul(x = var_3948_cast_fp16_0, y = layers_7_post_attention_layernorm_weight_promoted_to_fp16)[name = string("attn_output_cast_fp16")]; tensor x_153_cast_fp16 = add(x = x_139_cast_fp16, y = attn_output_cast_fp16)[name = string("x_153_cast_fp16")]; int32 var_3957 = const()[name = string("op_3957"), val = int32(-1)]; fp16 const_91_promoted_to_fp16 = const()[name = string("const_91_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3959_cast_fp16 = mul(x = x_153_cast_fp16, y = const_91_promoted_to_fp16)[name = string("op_3959_cast_fp16")]; bool input_227_interleave_0 = const()[name = string("input_227_interleave_0"), val = bool(false)]; tensor input_227_cast_fp16 = concat(axis = var_3957, interleave = input_227_interleave_0, values = (x_153_cast_fp16, var_3959_cast_fp16))[name = string("input_227_cast_fp16")]; tensor normed_217_axes_0 = const()[name = string("normed_217_axes_0"), val = tensor([-1])]; fp16 var_3954_to_fp16 = const()[name = string("op_3954_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_217_cast_fp16 = layer_norm(axes = normed_217_axes_0, epsilon = var_3954_to_fp16, x = input_227_cast_fp16)[name = string("normed_217_cast_fp16")]; tensor var_3964_split_sizes_0 = const()[name = string("op_3964_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3964_axis_0 = const()[name = string("op_3964_axis_0"), val = int32(-1)]; tensor var_3964_cast_fp16_0, tensor var_3964_cast_fp16_1 = split(axis = var_3964_axis_0, split_sizes = var_3964_split_sizes_0, x = normed_217_cast_fp16)[name = string("op_3964_cast_fp16")]; tensor layers_7_pre_feedforward_layernorm_weight_promoted_to_fp16 = const()[name = string("layers_7_pre_feedforward_layernorm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(155279872)))]; tensor var_3966_cast_fp16 = mul(x = var_3964_cast_fp16_0, y = layers_7_pre_feedforward_layernorm_weight_promoted_to_fp16)[name = string("op_3966_cast_fp16")]; tensor gate_29 = linear(bias = linear_5_bias_0, weight = layers_7_mlp_gate_proj_weight_palettized, x = var_3966_cast_fp16)[name = string("linear_68")]; tensor up = linear(bias = linear_5_bias_0, weight = layers_7_mlp_up_proj_weight_palettized, x = var_3966_cast_fp16)[name = string("linear_69")]; string gate_mode_0 = const()[name = string("gate_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gate = gelu(mode = gate_mode_0, x = gate_29)[name = string("gate")]; tensor input_231 = mul(x = gate, y = up)[name = string("input_231")]; tensor x_155 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_7_mlp_down_proj_weight_palettized, x = input_231)[name = string("linear_70")]; int32 var_3988 = const()[name = string("op_3988"), val = int32(-1)]; fp16 const_92_promoted = const()[name = string("const_92_promoted"), val = fp16(-0x1p+0)]; tensor var_3990 = mul(x = x_155, y = const_92_promoted)[name = string("op_3990")]; bool input_233_interleave_0 = const()[name = string("input_233_interleave_0"), val = bool(false)]; tensor input_233 = concat(axis = var_3988, interleave = input_233_interleave_0, values = (x_155, var_3990))[name = string("input_233")]; tensor normed_221_axes_0 = const()[name = string("normed_221_axes_0"), val = tensor([-1])]; fp16 var_3985_to_fp16 = const()[name = string("op_3985_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_221_cast_fp16 = layer_norm(axes = normed_221_axes_0, epsilon = var_3985_to_fp16, x = input_233)[name = string("normed_221_cast_fp16")]; tensor var_3995_split_sizes_0 = const()[name = string("op_3995_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_3995_axis_0 = const()[name = string("op_3995_axis_0"), val = int32(-1)]; tensor var_3995_0, tensor var_3995_1 = split(axis = var_3995_axis_0, split_sizes = var_3995_split_sizes_0, x = normed_221_cast_fp16)[name = string("op_3995")]; tensor hidden_states_59 = mul(x = var_3995_0, y = layers_7_post_feedforward_layernorm_weight)[name = string("hidden_states_59")]; tensor hidden_states_61_cast_fp16 = add(x = x_153_cast_fp16, y = hidden_states_59)[name = string("hidden_states_61_cast_fp16")]; tensor per_layer_slice_begin_0 = const()[name = string("per_layer_slice_begin_0"), val = tensor([0, 0, 1792])]; tensor per_layer_slice_end_0 = const()[name = string("per_layer_slice_end_0"), val = tensor([1, 8, 2048])]; tensor per_layer_slice_end_mask_0 = const()[name = string("per_layer_slice_end_mask_0"), val = tensor([true, true, false])]; tensor per_layer_slice_cast_fp16 = slice_by_index(begin = per_layer_slice_begin_0, end = per_layer_slice_end_0, end_mask = per_layer_slice_end_mask_0, x = per_layer_combined_out)[name = string("per_layer_slice_cast_fp16")]; tensor gated_29 = linear(bias = linear_2_bias_0, weight = layers_7_per_layer_input_gate_weight_palettized, x = hidden_states_61_cast_fp16)[name = string("linear_71")]; string gated_mode_0 = const()[name = string("gated_mode_0"), val = string("TANH_APPROXIMATION")]; tensor gated = gelu(mode = gated_mode_0, x = gated_29)[name = string("gated")]; tensor input_237_cast_fp16 = mul(x = gated, y = per_layer_slice_cast_fp16)[name = string("input_237_cast_fp16")]; tensor layers_7_per_layer_projection_weight_promoted_to_fp16_palettized = constexpr_lut_to_dense(indices = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(155283008))), lut = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(155479680))))[name = string("layers_7_per_layer_projection_weight_promoted_to_fp16_palettized")]; tensor linear_72_cast_fp16 = linear(bias = linear_4_bias_0_to_fp16, weight = layers_7_per_layer_projection_weight_promoted_to_fp16_palettized, x = input_237_cast_fp16)[name = string("linear_72_cast_fp16")]; int32 var_4033 = const()[name = string("op_4033"), val = int32(-1)]; fp16 const_93_promoted_to_fp16 = const()[name = string("const_93_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4035_cast_fp16 = mul(x = linear_72_cast_fp16, y = const_93_promoted_to_fp16)[name = string("op_4035_cast_fp16")]; bool input_interleave_0 = const()[name = string("input_interleave_0"), val = bool(false)]; tensor input_cast_fp16 = concat(axis = var_4033, interleave = input_interleave_0, values = (linear_72_cast_fp16, var_4035_cast_fp16))[name = string("input_cast_fp16")]; tensor normed_225_axes_0 = const()[name = string("normed_225_axes_0"), val = tensor([-1])]; fp16 var_4030_to_fp16 = const()[name = string("op_4030_to_fp16"), val = fp16(0x1.1p-20)]; tensor normed_225_cast_fp16 = layer_norm(axes = normed_225_axes_0, epsilon = var_4030_to_fp16, x = input_cast_fp16)[name = string("normed_225_cast_fp16")]; tensor var_4040_split_sizes_0 = const()[name = string("op_4040_split_sizes_0"), val = tensor([1536, 1536])]; int32 var_4040_axis_0 = const()[name = string("op_4040_axis_0"), val = int32(-1)]; tensor var_4040_cast_fp16_0, tensor var_4040_cast_fp16_1 = split(axis = var_4040_axis_0, split_sizes = var_4040_split_sizes_0, x = normed_225_cast_fp16)[name = string("op_4040_cast_fp16")]; tensor layers_7_post_per_layer_input_norm_weight_promoted_to_fp16 = const()[name = string("layers_7_post_per_layer_input_norm_weight_promoted_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(155481280)))]; tensor hidden_states_63_cast_fp16 = mul(x = var_4040_cast_fp16_0, y = layers_7_post_per_layer_input_norm_weight_promoted_to_fp16)[name = string("hidden_states_63_cast_fp16")]; tensor hidden_states_cast_fp16 = add(x = hidden_states_61_cast_fp16, y = hidden_states_63_cast_fp16)[name = string("hidden_states_cast_fp16")]; tensor const_94_promoted_to_fp16 = const()[name = string("const_94_promoted_to_fp16"), val = tensor([0x1.38p-1])]; tensor hidden_states_out = mul(x = hidden_states_cast_fp16, y = const_94_promoted_to_fp16)[name = string("op_4050_cast_fp16")]; } -> (hidden_states_out, per_layer_combined_out); }